Igneum Miner 0.3.11: program class v3 (Counter ASIC 2.0) and proving v1 on the devnet

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-josh 2026-10-06 01:12:49 +01:00
commit 30cd292cf0
647 changed files with 99978 additions and 398 deletions

View file

@ -71,8 +71,14 @@ jobs:
run: bash tools/ci/copied-sources-check.sh
- name: the signer is never piped into head
run: bash tools/ci/signer-pipe-check.sh
- name: bash bodies in PowerShell job scripts pass bash -n, the lost-quote class (self-test first, then the tree)
run: bash tools/ci/bash-body-check.sh --self-test && bash tools/ci/bash-body-check.sh
- name: run jobs test their fetched kit before use, the wiped-jobs-folder class (self-test first, then the tree)
run: bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh
- name: pinned guest programs match their manifest and are built only by pin-guests.sh
run: bash tools/ci/pinned-guests-check.sh
- name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026)
run: bash tools/ci/prover-socket-check.sh
- name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree)
run: bash tools/ci/no-secrets-check.sh --self-test && bash tools/ci/no-secrets-check.sh
- name: faucet unit tests (validation, the daily limits, the signed transaction; keccak, RLP and secp256k1 vectors)

View file

@ -219,7 +219,7 @@ dependencies = [
[[package]]
name = "igneum-app"
version = "0.3.10"
version = "0.3.11"
dependencies = [
"ed25519-dalek",
"getrandom",

View file

@ -1,6 +1,6 @@
[package]
name = "igneum-app"
version = "0.3.10"
version = "0.3.11"
edition = "2021"
description = "Igneum Miner engine: supervises the node, the miner and the GPU workers, and serves the dashboard on 127.0.0.1"
license = "MIT"

View file

@ -6,8 +6,8 @@
1 ICON "igneum.ico"
1 VERSIONINFO
FILEVERSION 0,3,10,0
PRODUCTVERSION 0,3,10,0
FILEVERSION 0,3,11,0
PRODUCTVERSION 0,3,11,0
FILEFLAGSMASK 0x3fL
FILEFLAGS 0x0L
FILEOS VOS_NT_WINDOWS32
@ -20,12 +20,12 @@ BEGIN
BEGIN
VALUE "CompanyName", "Igneum"
VALUE "FileDescription", "Igneum Miner engine"
VALUE "FileVersion", "0.3.10"
VALUE "FileVersion", "0.3.11"
VALUE "InternalName", "igneum-app"
VALUE "LegalCopyright", "Igneum contributors"
VALUE "OriginalFilename", "igneum-app.exe"
VALUE "ProductName", "Igneum Miner"
VALUE "ProductVersion", "0.3.10"
VALUE "ProductVersion", "0.3.11"
END
END
BLOCK "VarFileInfo"

View file

@ -87,6 +87,10 @@ pub struct Settings {
/// so it includes proof records it never verified (src/verifier.rs). Default off; a found verifier always wins.
#[serde(default)]
pub proof_verify_trust: bool,
/// Proving v1 step 1 (5 October 2026): the install-time default for `prove` has been applied once (src/provedefault.rs:
/// on when the machine can prove, never switching an explicit on back off). Older installs apply it at their next start.
#[serde(default)]
pub prove_default_applied: bool,
}
fn one() -> u32 {
@ -98,7 +102,7 @@ fn yes() -> bool {
impl Default for Settings {
fn default() -> Settings {
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false }
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false }
}
}

View file

@ -983,3 +983,19 @@ mod tests {
assert_eq!(big.identities, 8);
}
}
/// The machine's RAM in MB (the prover default's RAM gate, src/provedefault.rs): Windows through
/// `Win32_OperatingSystem.TotalVisibleMemorySize` (KB), Linux through `/proc/meminfo`, macOS through `sysctl hw.memsize`;
/// None when unreadable (no gate).
pub fn total_ram_mb() -> Option<u64> {
if cfg!(windows) {
let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "(Get-CimInstance Win32_OperatingSystem).TotalVisibleMemorySize"]), None, Duration::from_secs(20))?;
return out.replace('\0', "").trim().parse::<u64>().ok().map(|kb| kb / 1024);
}
if cfg!(target_os = "linux") {
let text = std::fs::read_to_string("/proc/meminfo").ok()?;
return text.lines().find(|l| l.starts_with("MemTotal:")).and_then(|l| l.split_whitespace().nth(1)).and_then(|kb| kb.parse::<u64>().ok()).map(|kb| kb / 1024);
}
let out = run_timeout(Command::new("sysctl").args(["-n", "hw.memsize"]), None, Duration::from_secs(5))?;
out.trim().parse::<u64>().ok().map(|b| b / (1024 * 1024))
}

View file

@ -253,6 +253,40 @@ impl Shared {
Ok(json!({ "ok": true, "address": w.address, "display": keys::checksum(&w.address), "private_key": w.private_key, "wallet_file": self.wallet_path.display().to_string() }))
}
/// Proving v1 step 1 (5 October 2026): once per install, after the cards are known, the prover goes on by itself
/// when this machine can prove (src/provedefault.rs: an NVIDIA card with 12 GB or more, WSL2 on Windows, Linux
/// native, Apple silicon off until measured). An explicit on is never switched off; the line goes to the log and
/// to the Proving tile. Older installs apply it at their first start on this version.
pub fn apply_prove_default(&self) {
let (applied, already_on) = {
let s = self.settings.lock().unwrap();
(s.prove_default_applied, s.prove)
};
if applied {
return;
}
let cards = self.state.lock().unwrap().mining.cards.clone();
let wsl = if cfg!(windows) { Some(crate::wslhost::distro_answers()) } else { None };
let d = crate::provedefault::decide(&cards, std::env::consts::OS, wsl, crate::detect::total_ram_mb());
let on = d.on || already_on;
{
let mut s = self.settings.lock().unwrap();
s.prove = on;
s.prove_default_applied = true;
s.save(&self.settings_path);
}
{
let mut st = self.state.lock().unwrap();
st.settings.prove = on;
st.proving.enabled = on;
st.proving.default_note = d.line.clone();
if !on {
st.proving.status = "off".into();
}
}
self.log(&format!("prover default: {}{}", d.line, if already_on && !d.on { " (left on: it was switched on by hand)" } else { "" }));
}
/// The prover service switch (src/prover.rs); the thread picks it up within 10 s.
pub fn set_prove(&self, on: bool) -> Result<Value, String> {
{
@ -415,6 +449,42 @@ fn jitter_secs(seed: u64) -> u64 {
5 + (seed.wrapping_mul(2654435761) >> 7) % 56
}
/// The `--prepare-packs` directory the miner gets, relative to the app data folder (its cwd), in the platform's
/// separator: `packs\prepare` on Windows, `packs/prepare` elsewhere. Every worker gets it (program class v3).
pub(crate) fn prepare_packs_arg() -> String {
if cfg!(windows) { "packs\\prepare".to_string() } else { "packs/prepare".to_string() }
}
/// Seconds after a resume before every enabled card must be mining (a worker takes 10 to 60 s to its first STATUS
/// line with a hash rate; the pack export before it a few seconds more).
pub(crate) const RESUME_CHECK_SECS: u64 = 90;
/// What the resume rule reads from a miner slot.
#[derive(Clone, Debug, PartialEq, Eq)]
pub(crate) struct ResumeSlot {
/// the watchdog marked the card faulted
pub faulted: bool,
/// a worker process is alive on the slot
pub live: bool,
}
/// The resume rule: every slot without a live worker is re-armed, faulted or not. (The rule before 5 October 2026
/// re-armed only faulted slots; `stop_miners("paused")` had cleared every slot's `restart_at`, so a healthy paused
/// card never restarted: PC 2 at 21:25:11Z, the Mac that afternoon.)
pub(crate) fn slots_to_rearm_on_resume(slots: &[ResumeSlot]) -> Vec<usize> {
slots.iter().enumerate().filter(|(_, s)| !s.live).map(|(i, _)| i).collect()
}
/// The check RESUME_CHECK_SECS after a resume: every enabled, present, non-faulted card must report a hash rate
/// above 0 or its name is returned with its state (the caller logs one line per card).
pub(crate) fn resume_check(cards: &[CardState]) -> Vec<String> {
cards
.iter()
.filter(|c| c.enabled && c.present() && c.state != "faulted" && c.hash_now <= 0.0)
.map(|c| format!("{} is not mining {RESUME_CHECK_SECS} s after resume (state {}, pid {}{})", c.name, c.state, c.pid, if c.message.is_empty() { String::new() } else { format!(", {}", c.message) }))
.collect()
}
struct MinerSlot {
card: usize,
label: String,
@ -443,6 +513,8 @@ pub struct Engine {
node: Option<Proc>,
node_external: bool,
node_restart_at: Option<Instant>,
/// resume rule (5 October 2026): when due, every enabled card must be mining or its name goes to the log
resume_check_at: Option<Instant>,
node_started_at: Instant,
node_starts: u32,
node_restarts: u32,
@ -545,6 +617,7 @@ impl Engine {
node: None,
node_external: false,
node_restart_at: None,
resume_check_at: None,
node_started_at: now,
node_starts: 0,
node_restarts: 0,
@ -711,6 +784,8 @@ impl Engine {
self.detect_again = false;
self.shared.send(Cmd::Detect);
}
// proving v1 step 1: the install-time prover default, once the cards are known
self.shared.apply_prove_default();
}
Cmd::ApplyCards(choices) => self.apply_cards(choices),
Cmd::Start => self.start(),
@ -729,13 +804,23 @@ impl Engine {
self.shared.save_settings();
self.st().mining.paused = false;
self.shared.event("ok", "mining resumed");
// a user action: faulted cards try again
for m in self.miners.iter_mut() {
// the resume rule (5 October 2026, PC 2 at 21:25:11Z and the Mac that afternoon): stop_miners("paused")
// cleared every slot's restart_at and this arm re-armed only FAULTED slots, so a card whose worker had
// simply been stopped stayed "off" at 0 MH/s until the app was relaunched. Now every slot without a
// live worker is re-armed (a faulted one reset first), its pack is exported again before the start
// (prepared = false: the hour may have turned while paused), and a check 90 s later names any
// enabled card that is not mining (`resume_check`).
let views: Vec<ResumeSlot> = self.miners.iter().map(|m| ResumeSlot { faulted: m.watch.faulted().is_some(), live: m.proc.is_some() }).collect();
let now = Instant::now();
for i in slots_to_rearm_on_resume(&views) {
let m = &mut self.miners[i];
if m.watch.faulted().is_some() {
m.watch.event(0.0, crate::watchdog::Event::Reset);
m.restart_at = Some(Instant::now());
}
m.restart_at = Some(now);
m.prepared = false;
}
self.resume_check_at = Some(now + Duration::from_secs(RESUME_CHECK_SECS));
if !self.running {
self.start();
}
@ -1385,13 +1470,16 @@ impl Engine {
a.push("--network".into());
a.push(r.network.clone());
}
// The miner runs with the app data folder as its cwd: the pack paths stay relative (the miner splits
// --worker-args on spaces, and %LOCALAPPDATA% may carry a space in the user name). Every worker gets
// --prepare-packs (5 October 2026, program class v3: the Metal worker too compiles v3 only from a prepared
// pack, `prepare <e> <d> <dir> class=v3 era=<hex>`; before this the flag went to the CUDA and OpenCL
// workers only and a Mac would answer `need` lines at the first v3 epoch and stop mining, the 18:23Z class).
a.push("--prepare-packs".into());
a.push(prepare_packs_arg());
if card.worker != "Metal" {
// The miner runs with the app data folder as its cwd: the pack paths stay relative (the miner splits
// --worker-args on spaces, and %LOCALAPPDATA% may carry a space in the user name). The prebuilt workers
// take the exported pack with --pack and build the next program from --prepare-packs; a worker built
// from source has the program compiled in and exits 42 at the boundary instead.
a.push("--prepare-packs".into());
a.push("packs\\prepare".into());
// The prebuilt workers take the exported pack with --pack and build the next program from
// --prepare-packs; a worker built from source has the program compiled in and exits 42 at the boundary.
if card.worker == "OpenCL" {
a.push("--job-nonces".into());
a.push("2097152".into());
@ -1426,7 +1514,9 @@ impl Engine {
self.miners[i].starts += 1;
let seg = if self.miners[i].starts > 1 { format!("-r{}", self.miners[i].starts) } else { String::new() };
let log = self.shared.runtime.log_dir.join(format!("miner-{}-{}{seg}.log", self.miners[i].label, self.stamp));
let cwd = if card.worker == "Metal" { None } else { Some(self.shared.runtime.app_dir.clone()) };
// every worker runs with the app data folder as its cwd, so the relative pack paths resolve (the Metal worker
// too, since it takes --prepare-packs now)
let cwd = Some(self.shared.runtime.app_dir.clone());
self.miners[i].prepared = false;
// the fleet's per-card kernel tuning (from the signed manifest) reaches the GPU worker through the miner's environment
let envs: Vec<(String, String)> = self.ota.tuning_path().map(|p| vec![("IGNEUM_TUNING_FILE".to_string(), p.display().to_string())]).unwrap_or_default();
@ -2157,6 +2247,18 @@ impl Engine {
self.tick_node(now);
self.tick_watch(now);
self.tick_miners(now);
if let Some(at) = self.resume_check_at {
if now >= at {
self.resume_check_at = None;
let (cards, paused) = { let st = self.st(); (st.mining.cards.clone(), st.mining.paused) };
if !paused {
for line in resume_check(&cards) {
self.shared.log(&format!("resume: {line}"));
self.shared.event("error", &format!("resume: {line}"));
}
}
}
}
self.tick_telemetry(now);
self.tick_sweep(now);
}
@ -3356,6 +3458,62 @@ pub fn parse_race(body: &str) -> Option<RaceParsed> {
Some(r)
}
#[cfg(test)]
mod prepare_packs_tests {
use super::*;
/// Program class v3 (5 October 2026): the Metal worker needs the prepare directory too, in the platform's
/// separator, relative to the app data folder the miner runs in.
#[test]
fn every_worker_gets_the_prepare_directory_in_the_platform_form() {
let arg = prepare_packs_arg();
if cfg!(windows) {
assert_eq!(arg, "packs\\prepare");
} else {
assert_eq!(arg, "packs/prepare", "the Mac's Metal worker gets a forward-slash path");
}
assert!(!arg.contains(' ') && !arg.starts_with('/'), "relative, no space: the miner splits --worker-args on spaces and the data folder may carry one");
}
}
#[cfg(test)]
mod resume_tests {
use super::*;
fn card(name: &str, enabled: bool, state: &str, hash: f64) -> CardState {
CardState { name: name.into(), vendor: "nvidia".into(), kind: "discrete".into(), enabled, state: state.into(), hash_now: hash, ..Default::default() }
}
/// The state machine: paused (every slot stopped, restart_at cleared) -> resumed -> every slot without a live
/// worker is re-armed, faulted or not.
#[test]
fn resume_rearms_every_stopped_slot() {
let slots = [ResumeSlot { faulted: false, live: false }, ResumeSlot { faulted: true, live: false }, ResumeSlot { faulted: false, live: true }];
assert_eq!(slots_to_rearm_on_resume(&slots), vec![0, 1], "the healthy stopped slot and the faulted one restart; the live one is left alone");
}
/// The known-failed case (PC 2, 5 October 2026, 21:25:11Z, app 0.3.9: `[ok] mining resumed`, then `0.00 MH/s,
/// waiting` for 20 minutes): one healthy slot, stopped by the pause, not faulted. The old rule re-armed only
/// faulted slots and returned nothing for it; the new rule returns it.
#[test]
fn the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old() {
let old_rule = |slots: &[ResumeSlot]| -> Vec<usize> { slots.iter().enumerate().filter(|(_, s)| s.faulted).map(|(i, _)| i).collect() };
let pc2 = [ResumeSlot { faulted: false, live: false }];
assert!(old_rule(&pc2).is_empty(), "the 0.3.9 rule left the 5090's slot unarmed: this is the defect");
assert_eq!(slots_to_rearm_on_resume(&pc2), vec![0]);
}
/// The check 90 s after a resume names every enabled card without a hash rate, and nothing else.
#[test]
fn resume_check_names_the_cards_not_mining() {
let cards = [card("NVIDIA GeForce RTX 5090", true, "off", 0.0), card("AMD Radeon(TM) Graphics", true, "mining", 3.4), card("Intel UHD", false, "off", 0.0), card("RTX 3060", true, "faulted", 0.0)];
let lines = resume_check(&cards);
assert_eq!(lines.len(), 1, "{lines:?}");
assert!(lines[0].starts_with("NVIDIA GeForce RTX 5090 is not mining 90 s after resume (state off"), "{}", lines[0]);
assert!(resume_check(&[card("RTX 5090", true, "mining", 118.9)]).is_empty());
}
}
#[cfg(test)]
mod tests {
use super::{digest_from_line, parse_race, switches_of, sync_decision, Reading};

View file

@ -29,6 +29,7 @@ mod jobs;
mod jobrun;
mod jobbuild;
mod prover;
mod provedefault;
mod verifier;
mod wslhost;
mod sweep;

View file

@ -0,0 +1,169 @@
//! Proving v1 step 1 (5 October 2026, Josh: "open the proving round asap"): the prover is on by default on every
//! mining machine that can prove, decided once per install after the cards are detected (src/engine.rs
//! `apply_prove_default`). The rule, one line each:
//!
//! | Machine | Default | Why (bench-log 5 October 2026, "proving v1", the S_p curve on the RTX 5090, SP1 6.8.1's GPU prover) |
//! |---|---|---|
//! | NVIDIA card with 24 GB or more, mining or not, Windows with WSL2 (Ubuntu-24.04) answering or Linux | on | a full shard at the adopted v1 budget (30,000 pgas, 4.7 M cycles) peaks at 20,434 MiB alone and 22,210 beside the miner (measured on the 5090; approximate for a 24 GB card's own allocation); the prototype shard the devnet proves until its fee switch (6.75 M pgas) peaks at 28,307 MiB alone and 30,039 beside the miner, so until the switch only a 32 GB card proves it and a 24 GB card's prover waits for shards it can hold (the host refuses nothing; a proof that runs out of memory fails and the shard is left) |
//! | NVIDIA card of 16 to 24 GB | off, with the line saying why | the GPU prover's floor is 13,874 MiB for an EMPTY shard, 15,670 beside the miner; a 16 GB card holds no full shard |
//! | NVIDIA card under 16 GB | off | 13,874 MiB does not fit; Josh's 12 GB requirement is open until a prover build with a smaller floor is measured |
//! | Windows under 32 GB of RAM | off, with the line saying why | the WSL2 prover held 7.9 GB on a 63 GB PC; a 16 GB PC would swap |
//! | Windows with a qualifying card but WSL2 silent | off, with the Set up hint | nothing can prove until the distribution exists |
//! | Apple silicon | off | the M5 Max CPU took 41 to 55 s for an EMPTY shard's compressed proof under load and 272 s for a 200-pgas shard; a full shard was never under 60 s (bench-log 4 and 5 October 2026) |
//! | AMD-only (no NVIDIA card) | off, "mines and does not prove" | no zkVM proves on an AMD GPU today (docs/analysis/amd-proving.md); the SP1 CPU prover on PC 1 cost 82 to 87 s core plus 199 to 202 s compressed a shard at a 30 GB RSS whatever the shard size (bench-log, "the SP1 CPU prover on PC 1") |
//!
//! Decided 5 October 2026 (delegated by Josh: "deploy what is absolute best"), docs/plans/proving-v1.md. The default
//! never switches an explicit on back off, and Settings always wins afterwards.
use crate::state::CardState;
/// A card that proves, mining or not, needs this much: the adopted v1 shard peaks at 20,434 MiB alone (the S_p curve,
/// 5 October 2026) and 22,210 beside the miner; `nvidia-smi` reports MiB and a 24 GB card reports 24,564, so the
/// test is at 23 GB. (The same value for a mining and an idle card: the floor is the GPU server's, not the miner's.)
pub const MIN_VRAM_MB_MINING: u64 = 23_552;
pub const MIN_VRAM_MB_PROVE_ONLY: u64 = 23_552;
/// What a 32 GB card alone can do that a 24 GB one cannot: the prototype shard (28,307 MiB alone, 30,039 beside the
/// miner), the devnet's shard until its fee switch at DAA 210,000; `nvidia-smi` reports 32,607 for the RTX 5090.
pub const VRAM_MB_PROTOTYPE_SHARD: u64 = 31_000;
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct Decision {
pub on: bool,
/// One plain sentence for the log and the Proving tile.
pub line: String,
}
fn gb(mb: u64) -> u64 {
(mb + 512) / 1024
}
/// Windows machines under this much RAM stay off until measured (consequences review C4, 5 October 2026): PC 2 at
/// 63 GB had 25.6 GB in use with the WSL2 VM's working set at 7.9 GB while proving; a 16 GB PC would swap.
pub const MIN_RAM_MB_WINDOWS: u64 = 31_000;
/// The card an aggregation (the chained SP1 recursion, spec 7.8) may run on: 16,751 MiB measured with the miner
/// resident (13.4 GB alone, approximate), so a 24 GB card mining or not; the same gate as the shard prover.
pub fn aggregation_card(cards: &[CardState]) -> Option<&CardState> {
cards.iter().filter(|c| c.vendor == "nvidia" && c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).max_by_key(|c| c.vram_mb)
}
/// `os` is `std::env::consts::OS` ("windows", "linux", "macos"); `wsl_answers` is read on Windows only; `ram_mb` is the
/// machine's RAM when the platform reports it (None = unknown, no gate).
pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb: Option<u64>) -> Decision {
let nvidia: Vec<&CardState> = cards.iter().filter(|c| c.vendor == "nvidia").collect();
// a mining card needs 20 GB (the measured mine-and-prove peak of 16.8 GB), a card that only proves 16 GB
let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).collect();
let off = |line: String| Decision { on: false, line };
if os == "macos" {
return off("proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)".into());
}
let Some(best) = able.iter().max_by_key(|c| c.vram_mb) else {
let seen = if nvidia.is_empty() {
"no NVIDIA card".to_string()
} else {
nvidia.iter().map(|c| format!("{} {} GB{}", c.name, gb(c.vram_mb), if c.enabled { ", mining" } else { "" })).collect::<Vec<_>>().join(", ")
};
let why = if nvidia.iter().any(|c| c.vram_mb >= 15_872) {
"a full shard needs a 24 GB card (measured 20.4 GB on the adopted shard size, 13.9 GB for an empty one); this card is under that, so Settings would switch proving on at your own risk"
} else if nvidia.is_empty() {
"this machine mines and does not prove: no zkVM proves on an AMD GPU today, and the CPU prover costs about 5 minutes a shard at a 30 GB RSS (bench-log, the SP1 CPU prover on PC 1); proving needs an NVIDIA card with 24 GB or more"
} else {
"no NVIDIA card with 24 GB or more (the GPU prover's floor is 13.9 GB for an empty shard and 20.4 GB for a full one)"
};
return off(format!("proving off by default: {why} ({seen})"));
};
let card = format!("{} ({} GB{})", best.name, gb(best.vram_mb), if best.enabled { ", mining too" } else { ", proving only" });
if os == "windows" {
if let Some(ram) = ram_mb {
if ram < MIN_RAM_MB_WINDOWS {
return off(format!("proving off by default: {card} qualifies but this PC has {} GB of RAM; proving needs 32 GB on Windows until a smaller PC is measured (the WSL2 prover held 7.9 GB on a 63 GB PC); Settings switches it on", gb(ram)));
}
}
}
let size_note = if best.vram_mb >= VRAM_MB_PROTOTYPE_SHARD { "" } else { "; until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" };
match os {
"windows" => match wsl_answers {
Some(true) => Decision { on: true, line: format!("proving on by default: {card} with WSL2 (Ubuntu-24.04 answers){size_note}; Settings switches it off") },
_ => off(format!("proving off: {card} qualifies but WSL2 (Ubuntu-24.04) did not answer; Set up installs it, then Settings switches proving on")),
},
"linux" => Decision { on: true, line: format!("proving on by default: {card} on Linux (the host runs next to the engine){size_note}; Settings switches it off") },
other => off(format!("proving off: {card} on {other}, no prover path there; Settings switches it on")),
}
}
#[cfg(test)]
mod tests {
use super::*;
fn card(vendor: &str, name: &str, vram_mb: u64) -> CardState {
CardState { vendor: vendor.into(), name: name.into(), vram_mb, enabled: true, ..Default::default() }
}
fn idle(vendor: &str, name: &str, vram_mb: u64) -> CardState {
CardState { enabled: false, ..card(vendor, name, vram_mb) }
}
#[test]
fn a_5090_with_wsl2_on_windows_is_on() {
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607), card("amd", "AMD Radeon(TM) Graphics", 512)], "windows", Some(true), Some(63_132));
assert!(d.on);
assert!(d.line.starts_with("proving on by default: NVIDIA GeForce RTX 5090 (32 GB, mining too) with WSL2"), "{}", d.line);
}
#[test]
fn windows_without_wsl2_is_off_with_the_setup_hint() {
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(false), Some(65_000));
assert!(!d.on);
assert!(d.line.contains("did not answer") && d.line.contains("Set up"), "{}", d.line);
assert!(!decide(&[card("nvidia", "RTX 4090", 24_564)], "windows", None, Some(65_000)).on, "an unread probe is not an answer");
}
#[test]
fn linux_needs_no_wsl2_and_the_memory_gates_hold() {
// the tiers of the S_p curve: 32 GB on with no note; 24 GB on with the prototype-size note; 16 GB and 12 GB off
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607)], "linux", None, None);
assert!(d.on && !d.line.contains("fee switch"), "{}", d.line);
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None);
assert!(d.on && d.line.contains("needs 32 GB, so this card proves from the switch on"), "{}", d.line);
assert!(decide(&[idle("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None).on, "mining or not, 24 GB proves");
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None);
assert!(!d.on);
assert!(d.line.contains("a full shard needs a 24 GB card") && d.line.contains("RTX 5080 16 GB, mining"), "{}", d.line);
assert!(!decide(&[idle("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None).on, "16 GB holds no full shard even alone");
let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 3060", 12_288)], "linux", None, None);
assert!(!d.on);
assert!(d.line.contains("no NVIDIA card with 24 GB or more") && d.line.contains("RTX 3060 12 GB"), "{}", d.line);
assert!(!decide(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None, None).on);
let d = decide(&[card("amd", "Radeon RX 9070 XT", 16_384)], "linux", None, None);
assert!(!d.on && d.line.contains("mines and does not prove"), "{}", d.line);
assert!(decide(&[], "linux", None, None).line.contains("mines and does not prove"));
}
#[test]
fn apple_silicon_stays_off() {
let d = decide(&[card("apple", "Apple M5 Max", 65_536)], "macos", None, Some(65_536));
assert!(!d.on);
assert!(d.line.contains("Apple silicon"));
assert!(!decide(&[card("nvidia", "RTX 5090", 32_607)], "macos", Some(true), None).on, "the OS rule comes first");
}
#[test]
fn a_windows_pc_under_32_gb_stays_off_and_the_aggregation_card_follows_the_same_gate() {
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), Some(16_300));
assert!(!d.on);
assert!(d.line.contains("16 GB of RAM") && d.line.contains("needs 32 GB on Windows"), "{}", d.line);
assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), None).on, "unknown RAM is not a gate");
assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, Some(16_300)).on, "the RAM gate is Windows only (the WSL2 VM)");
let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4070 Ti", 12_282)];
assert!(aggregation_card(&cards).is_none(), "a mining 16 GB card and an idle 12 GB card cannot aggregate");
let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4090", 24_564)];
assert_eq!(aggregation_card(&cards).map(|c| c.name.as_str()), Some("RTX 4090"));
let cards = [card("nvidia", "RTX 5090", 32_607)];
assert_eq!(aggregation_card(&cards).map(|c| c.vram_mb), Some(32_607));
}
#[test]
fn the_biggest_qualifying_card_is_named() {
let d = decide(&[idle("nvidia", "RTX 4090", 24_564), card("nvidia", "RTX 5090", 32_607)], "linux", None, None);
assert!(d.line.contains("RTX 5090 (32 GB, mining too)"), "{}", d.line);
}
}

View file

@ -339,7 +339,9 @@ pub fn start(shared: Arc<Shared>, bin_dir: PathBuf) {
fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
let mut attempted: HashSet<(String, u32)> = HashSet::new();
let mut attempted_segments: HashSet<u64> = HashSet::new();
let mut tools: Option<Tools> = None;
let ram_mb = crate::detect::total_ram_mb();
let mut last_probe = Instant::now() - Duration::from_secs(600);
let mut submitted: Vec<(u64, String, u32, u128)> = Vec::new();
let mut last_verifier_read = Instant::now() - Duration::from_secs(600);
@ -406,6 +408,20 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
}
}
let Some(t) = tools.as_ref() else { continue };
// consequences review C22 (5 October 2026): the SP1 CPU prover takes 29.5 to 30.5 GB of RSS and about five
// minutes a shard whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1"); on a machine under
// 32 GB it would swap the node out, so the CPU path is refused here, Settings or not, with the reason
if !t.cuda {
if let Some(ram) = ram_mb {
if ram < crate::provedefault::MIN_RAM_MB_WINDOWS {
set(&shared, |p| {
p.status = "off".into();
p.message = format!("the CPU prover needs 32 GB of RAM (30 GB measured on PC 1); this machine has {} GB, so proving stays off here", (ram + 512) / 1024);
});
continue;
}
}
}
set(&shared, |p| {
p.enabled = true;
p.available = true;
@ -466,6 +482,20 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
p.assigned = assigned;
p.keys = keys.len() as u32;
});
// proving v1 (spec 7.8): the aggregator step, when the node says v1 is active; one attempt a pass
if let Some((label0, _)) = keys.first() {
match aggregate_once(&shared, t, label0, &payout_address(&shared), &mut attempted_segments) {
Ok(Some(msg)) => {
shared.log(&format!("aggregator: {msg}"));
set(&shared, |p| p.segment_note = msg);
}
Ok(None) => {}
Err(e) => {
shared.log(&format!("aggregator: {e}"));
set(&shared, |p| p.segment_note = e);
}
}
}
let Some(w) = choose(&work, &attempted) else {
set(&shared, |p| {
p.status = if submitted.is_empty() { "idle".into() } else { "submitted".into() };
@ -505,11 +535,15 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
if !ok || !fixture.exists() {
return Err(format!("exporter: {}", out.lines().rev().find(|l| !l.trim().is_empty()).unwrap_or("failed")));
}
set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "proving on the CPU (slow)".into() });
set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "CPU prover: about five minutes a shard, 30 GB of RAM, paid only when no card proves first".into() });
let prover_env = if t.cuda { "cuda" } else { "cpu" };
let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &[("SP1_PROVER", prover_env), ("RUST_LOG", "off")], Duration::from_secs(3 * 3600), &dir.join(format!("prove-{}-{}.log", w.number, w.shard)));
if !ok || !results.exists() {
return Err(format!("prover: {}", out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed")));
let last = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string();
// the root-socket class (5 October 2026, PC 2 at 20:00Z and 21:25Z): a job that ran the host as root
// inside WSL2 left /tmp/sp1-cuda-0.sock owned by root, and this user's client cannot open it
let hint = if last.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user, left by a job that ran the prover as root: remove it as that user, or run the socket-fix job)" } else { "" };
return Err(format!("prover: {last}{hint}"));
}
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
let statement = res["statement"].as_str().ok_or("no statement in the results")?.to_string();
@ -558,6 +592,124 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
}
}
/// Proving v1 (spec 7.8): one aggregation attempt. When the node reports v1 active, takes the newest executed
/// segment that is still pending and not yet attempted here, needs one shard proof per shard of every block in
/// this node's pool (`igneum_getProofBytes`, a verified one when there is one) and, when the previous segment is
/// proven, its aggregated proof (`igneum_getSegmentProofBytes`); runs `igneum-prove-host --mode aggregate` over the
/// run of blocks (one process, one key setup), checks the public values against the node's native statement
/// (every field but `provers`), signs the record with the first key's label and submits it. Returns a line for
/// the log and the tile, or None when there is nothing to do.
fn aggregate_once(shared: &Shared, t: &Tools, label: &str, payout: &str, attempted: &mut HashSet<u64>) -> Result<Option<String>, String> {
let hexu = |x: &Value| x.as_str().and_then(|s| u64::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0);
let st = evm_rpc(shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10))?;
let v1 = &st["v1"];
if !v1["active"].as_bool().unwrap_or(false) || v1["start"].is_null() {
return Ok(None);
}
// the card gate (consequences review C2): a chained aggregation peaked at 16,751 MiB with the miner resident; the
// prover's own 24 GB gate applies; without such a card this machine proves shards and never aggregates
{
let cards = shared.state.lock().unwrap().mining.cards.clone();
if crate::provedefault::aggregation_card(&cards).is_none() {
return Ok(Some("no aggregation on this machine: it needs a 24 GB card (16.8 GB measured with the miner resident); shards still prove".into()));
}
}
let tip = hexu(&evm_rpc(shared, "eth_blockNumber", json!([]), Duration::from_secs(10))?);
let mut seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{tip:#x}")]), Duration::from_secs(10))?;
if !seg["executed"].as_bool().unwrap_or(false) {
let first = hexu(&seg["first"]);
if first == 0 || first - 1 < hexu(&v1["start"]) {
return Ok(None);
}
seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{:#x}", first - 1)]), Duration::from_secs(10))?;
}
let (first, last) = (hexu(&seg["first"]), hexu(&seg["last"]));
if seg["status"]["status"].as_str() != Some("pending") || attempted.contains(&first) || payout.len() != 42 {
return Ok(None);
}
let dir = shared.runtime.app_dir.join("proving").join(format!("seg-{first}"));
let _ = std::fs::create_dir_all(&dir);
let as_host_path = |p: &Path| if t.wsl { wsl_path(p) } else { p.display().to_string() };
// the shard proofs, one per shard of every block, from this node's pool
let mut groups: Vec<String> = Vec::new();
let mut missing: Vec<String> = Vec::new();
for b in seg["blocks"].as_array().cloned().unwrap_or_default() {
let n = hexu(&b["number"]);
let shards = b["shards"].as_u64().unwrap_or(0) as u32;
let have = b["shardProofs"].as_array().cloned().unwrap_or_default();
let mut files = Vec::new();
for i in 0..shards {
let pick = have.iter().find(|e| e["shard"].as_u64() == Some(i as u64) && e["verified"] == json!(true)).or_else(|| have.iter().find(|e| e["shard"].as_u64() == Some(i as u64)));
let Some(e) = pick else {
missing.push(format!("{n}/{i}"));
continue;
};
let got = evm_rpc(shared, "igneum_getProofBytes", json!([format!("{n:#x}"), i, e["keyHash"]]), Duration::from_secs(60))?;
let hex = got["proof"].as_str().ok_or("no proof bytes")?.trim_start_matches("0x").to_string();
let bytes: Vec<u8> = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect();
let f = dir.join(format!("b{n}-s{i}.bin"));
std::fs::write(&f, bytes).map_err(|e| e.to_string())?;
files.push(as_host_path(&f));
}
groups.push(files.join(","));
}
if !missing.is_empty() {
return Ok(Some(format!("segment {first}..{last}: waiting for shard proofs {} in this node's pool", missing.join(" "))));
}
// the previous segment's aggregated proof, when the chain continues
let prev = &seg["previous"];
let (prev_file, expected_pv) = if prev.is_null() {
(None, seg["publicValuesFresh"].as_str().unwrap_or("").to_string())
} else if prev["proofInPool"] == json!(true) {
let got = evm_rpc(shared, "igneum_getSegmentProofBytes", json!([prev["first"], prev["keyHash"]]), Duration::from_secs(60))?;
let hex = got["proof"].as_str().ok_or("no segment proof bytes")?.trim_start_matches("0x").to_string();
let bytes: Vec<u8> = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect();
let f = dir.join("prev-aggregated.bin");
std::fs::write(&f, bytes).map_err(|e| e.to_string())?;
(Some(as_host_path(&f)), seg["publicValuesContinuing"].as_str().unwrap_or("").to_string())
} else {
return Ok(Some(format!("segment {first}..{last}: the previous segment's proof is not in this node's pool; waiting")));
};
attempted.insert(first);
let parent = seg["blocks"][0]["parentHash"].as_str().ok_or("no parent hash")?.to_string();
let last_hash = seg["blocks"].as_array().and_then(|a| a.last()).and_then(|b| b["hash"].as_str()).ok_or("no last hash")?.to_string();
let results = dir.join("results.json");
let started = Instant::now();
set(shared, |p| p.message = format!("aggregating segment {first}..{last} ({})", if t.cuda { "GPU" } else { "CPU, slow" }));
let mut args: Vec<String> = vec!["--mode".into(), "aggregate".into(), "--proofs".into(), groups.join(";"), "--parent".into(), parent, "--out".into(), as_host_path(&results)];
if let Some(pf) = prev_file {
args.push("--prev".into());
args.push(pf);
}
let (ok, out) = run_tool(shared, t, &t.host, &args, &[("SP1_PROVER", if t.cuda { "cuda" } else { "cpu" }), ("RUST_LOG", "off")], Duration::from_secs(2 * 3600), &dir.join("aggregate.log"));
if !ok || !results.exists() {
return Err(format!("segment {first}..{last}: aggregator: {}", out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed")));
}
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
let pv = res["segment_public_values"].as_str().ok_or("no public values in the results")?.to_string();
let proof_sha = res["segment_proof_sha256"].as_str().ok_or("no proof hash in the results")?.to_string();
let proof_file = res["segment_proof_file"].as_str().ok_or("no proof file in the results")?.to_string();
// the node's native statement, every field but provers (bytes 236..268 of the 340)
let strip = |h: &str| { let h = h.trim_start_matches("0x"); if h.len() == 680 { format!("{}{}", &h[..472], &h[536..]) } else { h.to_string() } };
if strip(&pv) != strip(&expected_pv) {
return Err(format!("segment {first}..{last}: the aggregated statement differs from the node's native statement (it would be vetoed); ours {} node {}", &pv[..66.min(pv.len())], &expected_pv[..66.min(expected_pv.len())]));
}
let proof_path = if t.wsl { PathBuf::from(proof_file.replace("/mnt/c/", "C:/")) } else { PathBuf::from(proof_file) };
let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-segment-record", label, &chain_name(shared), &first.to_string(), &last.to_string(), &last_hash, payout, &pv, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-segment-record did not run")?;
let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-segment-record: {}", sg.trim()))?;
let record = signed["record"].as_str().ok_or("sign-segment-record gave no record")?.to_string();
let proof = std::fs::read(&proof_path).map_err(|e| format!("proof file {}: {e}", proof_path.display()))?;
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
let r = evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?;
if !r["accepted"].as_bool().unwrap_or(false) {
return Err(format!("segment {first}..{last}: record refused: {}", r["reason"].as_str().unwrap_or("?")));
}
let secs = started.elapsed().as_secs_f64();
shared.event("proving", &format!("segment {first}..{last} aggregated and submitted in {secs:.0} s (chain_len {})", res["segment_chain_len"]));
set(shared, |p| p.aggregated += 1);
Ok(Some(format!("segment {first}..{last} aggregated in {secs:.0} s and submitted; paid when a block carries it")))
}
/// Windows: runs the WSL2 setup from the payload (`wsl2/setup-wsl.sh` next to the engine) in a window of its own;
/// the user watches it and reboots when it asks. Elsewhere there is nothing to set up.
pub fn setup(shared: &Shared) -> Result<Value, String> {

View file

@ -196,6 +196,11 @@ pub struct ProvingState {
/// the pinned guests' ids (`igneum-prove-host --mode id`): the shard program and the aggregator; empty until read
pub program_id: String,
pub aggregator_id: String,
/// proving v1 step 1: the install-time default's one plain line (why proving is on or off on this machine)
pub default_note: String,
/// proving v1: segment records this machine aggregated and submitted, and the aggregator's last line
pub aggregated: u32,
pub segment_note: String,
}
#[derive(Clone, Serialize, Default)]

View file

@ -41,6 +41,43 @@ pub fn candidates(bin_dir: &Path) -> Vec<String> {
}
/// The candidates as one line for a message.
/// Whether the distribution answers at all (`wsl.exe -d Ubuntu-24.04 -- echo <marker>` within 30 s): the install-time
/// prover default (src/provedefault.rs) needs WSL2 on Windows before it switches proving on. Elsewhere: false.
#[allow(dead_code)] // also compiled into src/bin/prove-verify.rs, which does not call it
pub fn distro_answers() -> bool {
if !cfg!(windows) {
return false;
}
// self-contained (this file is also compiled into src/bin/prove-verify.rs, which has no detect or platform module)
let mut cmd = std::process::Command::new("wsl");
cmd.args(["-d", DISTRO, "--", "echo", "igneum-wsl-answers"]).stdin(std::process::Stdio::null()).stdout(std::process::Stdio::piped()).stderr(std::process::Stdio::null());
#[cfg(windows)]
{
use std::os::windows::process::CommandExt;
cmd.creation_flags(0x0800_0000); // CREATE_NO_WINDOW
}
let Ok(mut child) = cmd.spawn() else { return false };
let Some(out) = child.stdout.take() else { return false };
let reader = std::thread::spawn(move || {
let mut s = String::new();
let _ = std::io::Read::read_to_string(&mut std::io::BufReader::new(out), &mut s);
s
});
let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30);
loop {
match child.try_wait() {
Ok(Some(_)) => break,
Ok(None) if std::time::Instant::now() < deadline => std::thread::sleep(std::time::Duration::from_millis(100)),
_ => {
let _ = child.kill();
let _ = child.wait();
break;
}
}
}
reader.join().map(|o| o.replace('\0', "").contains("igneum-wsl-answers")).unwrap_or(false)
}
pub fn candidates_text(bin_dir: &Path) -> String {
candidates(bin_dir).join(", ")
}

View file

@ -1005,7 +1005,8 @@ if (typeof document !== 'undefined') (function () {
setText('pv-state', w.word + (pv.backend && enabled && pv.available ? ' (' + pv.backend.toUpperCase() + ')' : ''));
setText('pv-state-sub', w.sub);
var cell = $('pv-state').parentNode; cell.classList.toggle('ok', w.tone === 'on' || w.tone === 'ok'); cell.classList.toggle('bad', w.tone === 'bad');
setText('pv-note', w.note);
// proving v1: when off by the install-time default, the default's own line says why (src/provedefault.rs)
setText('pv-note', (!enabled && pv.default_note) ? 'Off. ' + pv.default_note + '.' : w.note);
setText('pv-assigned', String(pv.assigned || 0));
setText('pv-submitted', String(pv.submitted || 0));
setText('pv-paid', String(pv.paid || 0));

View file

@ -206,7 +206,7 @@
<div class="lead-row">
<div class="lead-text">
<h3>Prove shards on this machine</h3>
<p class="help">Every block on Igneum is turned into a short mathematical proof, in pieces called shards. The chain assigns shards to your keys; this machine proves them and is paid for each one.</p>
<p class="help">Every block on Igneum is turned into a short mathematical proof, in pieces called shards. The chain assigns shards to your keys; this machine proves them and is paid for each one. On by default on an NVIDIA card with 24 GB or more (a full shard needs 20.4 GB of GPU memory, measured); off on a Mac, whose CPU prover is slow.</p>
</div>
<label class="switch lg" title="Prove assigned shards"><input type="checkbox" id="s-prove" aria-label="Prove shards on this machine"><span class="track"></span></label>
</div>

View file

@ -3,6 +3,6 @@
// packaging/windows/Igneum-Miner.iss when the app version moves. Include guards, not #pragma once: rc.exe reads it too.
#ifndef IGNEUM_HOST_VERSION_H
#define IGNEUM_HOST_VERSION_H
#define IGNEUM_HOST_VERSION_STR "0.3.10"
#define IGNEUM_HOST_VERSION_RC 0,3,10,0
#define IGNEUM_HOST_VERSION_STR "0.3.11"
#define IGNEUM_HOST_VERSION_RC 0,3,11,0
#endif

View file

@ -0,0 +1,131 @@
# Proving on AMD and Apple cards: what exists, what the CPU can do, what to tell the public
5 October 2026, from Josh's two questions that evening: "test proving on the amd card?" and "can we test proving on
mac?". PC 1 holds an RTX 5090 and an RX 9070 XT (gfx1201, 16 GB) in an eGPU; this Mac is an M5 Max. The prover is
SP1 (`proving/igneum-prove`, `docs/plans/proving-v0.md`, `proving-v1.md`), run on the GPU only through SP1's CUDA
server. Every figure below is measured (with its bench-log entry or job id) or cited (with its file or page); the
rest is labelled approximate. Status words follow `docs/spec/00-overview.md` 0.2.
**The answer in three lines.** No zkVM proves on an AMD GPU on 5 October 2026: not SP1, not RISC Zero, not Jolt, not
OpenVM, and the ICICLE library underneath them has no AMD backend either. Apple silicon has a shipped Metal prover in
RISC Zero and a Metal backend in ICICLE, but SP1, the prover Igneum runs, is CPU-only on a Mac. So an AMD-only or
Apple-only machine mines and does not prove on its card; it can prove on its CPU, at the times measured in section 2.
## 1. The backends (read 5 October 2026, 20:30 to 20:50 UTC)
| Prover | Version read | CPU | NVIDIA (CUDA) | AMD (ROCm or HIP) | Apple (Metal) | Vulkan or WebGPU | Where it says so |
|---|---|---|---|---|---|---|---|
| SP1 (ours) | v6.8.1, 24 Sep 2026 (pinned); `dev` head 318dd530, 28 Sep 2026 | yes; AVX2 and AVX-512 on x86 through Plonky3 | yes: `sp1-gpu-server`, "Compute Capability 8.0 or higher", "24GB or more VRAM", "the CUDA 12 runtime and a compatible NVIDIA driver", Linux x86_64 | **no** | **no** | **no** | docs.succinct.xyz, SP1 docs "Hardware acceleration" page; `crates/sdk/src/lib.rs` (`pub mod cpu`, `mock`, `light`, `#[cfg(feature = "cuda")] pub mod cuda`, `#[cfg(feature = "network")] pub mod network`: no other backend module); `sp1-gpu/README.md` (`CUDA_ARCHS` 89, 90, 100, 120; NTT by NVIDIA cuPQC or sppark); release notes v6.2.3 to v6.8.1 (the only backend line: "add optional cuPQC NTT backend", v6.8.0); a code search of the repository on 5 October: "rocm" 0 files, "metal" 0, "vulkan" 0, "webgpu" 0; `cuobjdump` of sp1-gpu-server 6.8.1: sm_80, 86, 89, 90, 100, 120 and compute_120 PTX, nothing else (`docs/bench-log.md`, "proving v1", 5 October 2026) |
| sppark (SP1's NTT fallback, vendored at `sp1-gpu/crates/sys/sppark`) | `main` README, read 5 October 2026 | | yes: "x86_64 with Nvidia's Volta+ GPU hardware platforms on Linux and Windows" | "A limited support for AMD's RDNA and CDNA GPUs is provided" (upstream README). SP1's tree carries no HIP build: the 0 "rocm" files above, and `sp1-gpu/crates/sys/sppark/util/gpu_t.cuh` is CUDA only | no | no | github.com/supranational/sppark README; the SP1 files named |
| RISC Zero | latest release v3.0.6, 17 Jul 2026 (a v5.0.0-rc.1 of 15 Jan 2026 is also on the releases page) | yes, "nearly any modern CPU (x86 or ARM)" | yes, "RISC Zero targets NVIDIA GPUs using the CUDA framework" | **no** ("rocm", "vulkan": 0 files in the repository) | **yes**: `metal = ["prove"]` in `risc0/zkvm/Cargo.toml`; kernels in `risc0/sys/kernels/zkp/metal/*.metal` (zk, fri, mix, sha); docs: "RISC Zero will use the integrated Metal compute cores" on Apple silicon. The Groth16 wrapper "only works on x86 architecture, and so Apple Silicon is currently unsupported (even via Docker)" | no | dev.risczero.com "Local proving"; `risc0/zkvm/Cargo.toml` features `cuda = [... risc0-zkp/cuda ...]`, `metal = ["prove"]` |
| Jolt (a16z) | v0.3.0-alpha, 1 Oct 2025; "Jolt is in alpha and is not suitable for production use" | yes, "state-of-the-art performance on CPU" | no | **no** | a **draft** PR #1733 (opened 3 Aug 2026, not merged): titled "feat: Metal GPU backend (Apple Silicon)" and marked experimental: 92 Metal kernels, "2.12x speedup at 2^20 scale" on an M4 mini and "3.20x vs same-binary CPU" on an M5 Max, "Apple Silicon + macOS only. No CI coverage" | no | github.com/a16z/jolt README and book (jolt.a16zcrypto.com); PR #1733 |
| OpenVM | v2.0.2, 14 Aug 2026 | yes | yes: `cuda-backend` (v1.4.2 notes), "Improves the Halo2 GPU prover" (v2.0.2) | **no** | **no** | no | github.com/openvm-org/openvm releases |
| ICICLE (Ingonyama; the GPU library behind several provers, not SP1) | v4.0.0, 11 Jul 2025 | yes (MIT) | yes, "CUDA (for NVIDIA GPUs)" | **no** backend listed | yes, "Metal (for Apple Silicon GPUs)"; both under a special licence with a free research licence | Vulkan in the build system (PR #735, merged Jan 2025) and a draft "Vulkan NTT" PR #1019 (Jul 2025, "still wip"); nothing installable | dev.ingonyama.com "Install GPU backend"; the releases page; the README ("backends ... are distributed under a special license") |
Said plainly: **on 5 October 2026 no zkVM proves on an AMD GPU.** The only AMD code in the whole chain is sppark's
limited HIP path, which SP1 does not build. For Apple silicon the answer is split: RISC Zero ships a Metal prover
and ICICLE a Metal backend; SP1, Jolt and OpenVM do not. Nothing read tonight names an AMD plan with a date.
## 2. The CPU fallback, measured
SP1's CPU prover is the path an AMD-only or Apple-only machine has today. Three machines, the same pinned guests
(shard program id `0x2b1a81cb...`, aggregator `0x474678f3...`, pinned 2026-10-05T16:20:38Z), `SP1_PROVER=cpu`,
`--mode shard --shard 0` (execute, core proof, compressed proof, each verified). The RTX 5090 rows are the reference.
| Fixture (SP1 cycles) | Stage | PC 1 CPU, miner running on both cards (job `cpu-prove-pc1-small2`) | Apple M5 Max CPU (4 October, loaded; bench-log) | RTX 5090 (bench-log) |
|---|---|---|---|---|
| block-56-transfers-3shards shard 0, 200 pgas (315 k) | core | 82.5 s, 7,310,257 B, verify 0.210 s | 83.1 s, 7,310,257 B | not run on the 5090; the nearest rows are block-78 below and an empty live shard: 7.0 to 7.7 s compressed with the miner on the card (5 Oct, `chain-pc2-pv1b`, `pv1c`) |
| | compressed | 199.2 s, 1,272,897 B, verify 0.035 s; 312 s wall for setup 22.8 s, execute 0.14 s, core, compressed | 272.3 s, 1,272,897 B | |
| | peak RSS, CPU | 29.5 GB peak RSS; 978% CPU (9.8 of 16 cores), user 2,516 s, system 537 s | not recorded | |
| block-78-increment, 2 transactions (626 k) | core | 87.0 s, 7,317,857 B, verify 0.209 s | 22.0 s, 7.3 MB (3 October, v0 guest) | 1.4 s (4 October, mining paused) |
| | compressed | 202.3 s, 1,272,897 B, verify 0.034 s; 322 s wall (setup 21.8 s) | 55.7 s, 1.27 MB | 2.7 s |
| | peak RSS, CPU | 30.5 GB peak RSS; 979% CPU, user 2,616 s, system 541 s | not recorded | |
| block-338-shard1, one shard at `S_p` (60.8 M) | core | **not run**, by the PC 1 scheduler's decision at 21:05Z (PC 1's time tonight belongs to the Counter ASIC 2.0 gates; the job `cpu-prove-pc1-sp`, script `tools/amd-prove/pc1-cpu-prove-sp.ps1`, is written and unpublished). Extrapolation, approximate: 60.8 M cycles is about 29 SP1 shards of 2^21 cycles where the small fixtures are one, so the core proof alone is about 29 x 80 s, 40 min, and the compressed recursion over 29 shard proofs adds hours; the floor from the 5090's own ratios (6x on core, 4x on compressed between block-78 and `S_p`) is 9 min core and 13 min compressed. Either way far outside every deadline | not run on the CPU (execute alone 6.9 s) | 8.3 s |
| | compressed | not run (see the core cell) | not run | 10.9 s with the card to itself (4 Oct); 33.0 s with the miner running (5 Oct, `memminer-pc2-pv1`); 7.3 to 7.7 s per EMPTY shard with the miner running (`chain-pc2-pv1c`) |
| | peak RSS, CPU | not run; at least the 30 GB of the small rows | | GPU peak 28,295 MiB alone, 30,039 MiB beside the miner |
PC 1: Windows 11, WSL2 Ubuntu 24.04 as root, 16 cores and 46,994 MB visible to the VM, the Igneum Miner app 0.3.9 mining on the RTX 5090 (89% mean utilisation through both runs, 59 to 70% minimum: the miner, untouched) and on the RX 9070 XT (not visible to nvidia-smi, mining through the app's OpenCL worker). Job `cpu-prove-pc1-small2`, 20:49:00Z to 20:59:49Z, 649 s wall including a 6-s warm build; the host built without the `cuda` feature from the hosted package `igneum-prove-wsl2-pv1b.zip`, `--mode id` the pinned pair. The first job, `cpu-prove-pc1-small` (20:44 to 20:46Z), built the host cold in 126 s and proved nothing: an apostrophe inside a single-quoted awk program ended the quote, bash refused the whole loop and the job reported exit 0. The class fix: `tools/amd-prove/check-job-bash.sh` runs `bash -n` on the bash body of a PowerShell job before it is published, and the job itself runs `bash -n` inside the distro before the run; both were shown to fire on the bad body and pass the fixed one. Host RAM in the VM: 968 MB used before, 2,351 MB after; the prover's own peak 29.5 to 30.5 GB.
Mac, fresh run tonight: not taken. The Mac measure lock was held from 20:31Z (a read-width `packbench` under `measure`, three build slots, then a 1,500-s proving-v1 network under `run`) and did not free inside the 10-minute window the coordinator set, so the Apple column is the 4 October rows (M5 Max, 18 cores, 64 GB, load 38 to 47, `nice -n 19`): the same host modes on the same fixture, under heavier load than PC 1 tonight. The Mac's RAM peak was not recorded on 4 October; PC 1's 30 GB says a Mac needs more than 32 GB for the CPU prover, which a 64 GB M5 Max has and a 16 or 24 GB Mac does not.
The deadlines a CPU proof has to fit (all in the spec and the v1 plan): the exclusive window of an assigned shard is
10 s of DAA time (spec 7.2 item 3; after it anyone may prove and be paid first); the litepaper promises the block's
proof "within about a minute"; the launch target is 20 to 60 s behind the tip; from proving v1 a segment nobody has
proven in `T` = 600 DAA s (10 min) pays nothing (`docs/plans/proving-v1.md`, decisions). So a CPU shard proof is
useful only if it lands inside 10 min and competitive only if it lands inside about a minute.
Reading. On PC 1 the CPU proof of the smallest shard (315 k cycles) and of the two-transaction block (631 k cycles) cost the same: 82.5 and 87.0 s core, 199.2 and 202.3 s compressed. Doubling the cycles added 4.5 s to the core proof and 3.1 s to the compressed one, so about 280 s of every CPU proof is fixed cost (the recursion that turns the core proof into the 1.27 MB compressed proof the chain carries), and no shard size removes it. Against the deadlines: 282 s a shard (core plus compressed, the client already set up, as the app's loop runs it) is 28x the 10-s assignment window, 4.7x the minute the litepaper promises, and inside the 600-s unproven deadline of v1 with 5 min to spare; but an NVIDIA card proves the same shard in 2.7 to 7.7 s, so a CPU prover only ever wins a shard that no card has taken in 10 minutes. The Mac's 83.1 and 272.3 s of 4 October have the same shape. The RAM peak of 29.5 to 30.5 GB is the second finding: the SP1 CPU prover does not fit a 16 GB machine at all, and WSL2 gives a Windows VM half the host's RAM by default, so the CPU path needs a 64 GB Windows PC or a 32 GB Linux or Mac machine. The `S_p` shard on the CPU can only be slower (the 5090 takes 6x longer at `S_p` than on block-78: 8.3 s against 1.4 s core); it was not run tonight (the scheduler kept PC 1 for the Counter ASIC 2.0 gates) and could not change the conclusion.
## 3. What this means for each tier (the every-number rule, CLAUDE.md 5 October 2026)
| Tier | Mines | Proves on the card | The 20% proving-pool share (spec 2.5) | What the software does today |
|---|---|---|---|---|
| AMD-only home miner, one card of 8, 12 or 16 GB (an RX 9070 XT is 16 GB), Windows or Linux | yes (OpenCL worker, `proto-opencl`; PC 1's 9070 XT mines on the devnet) | **no**: no prover exists for the card | **lost**, unless CPU proving at a small shard size becomes a tier (section 4a) | the rig installer: `prover_decision` in `packaging/linux/bin/igneum-rig-lib.sh` (branch `rig-install`) skips every non-NVIDIA card (`[[ "$vendor" == nvidia ]] \|\| continue`) and prints "proving off by default: no NVIDIA card (no CUDA prover for AMD or Intel yet)"; the app: `provedefault.rs` (branch `proving-v1`) considers NVIDIA cards only. Both already right; neither offers the CPU path |
| Apple silicon (M-series, unified memory) | yes: the M5 Max at 26.7 MH/s (bench-log 4 October, "first hourly program swap", Metal `prepare 1` row) | **no** with SP1; RISC Zero and ICICLE have Metal, SP1 does not | **lost** today; a Metal prover behind the swappable interface would restore it (section 4b) | `provedefault.rs`: "proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)". Right |
| Mixed rig (NVIDIA and AMD cards in one box) | every card | the NVIDIA cards prove for the box; the AMD cards mine | kept, earned by the NVIDIA cards | the rig installer picks the biggest NVIDIA card (`prover_decision`, `PROVER_CARD` overrides), pauses its miner under 20 GB, keeps it mining at 20 GB or more; the AMD cards get a miner unit each. **The prover unit must never select an AMD card**: it does not (the vendor filter above), and that filter is now a stated requirement, not an accident |
| NVIDIA home miner, 8 or 12 GB | yes | no on this SP1 build (13.9 GB floor on an empty shard, `memsweep-pc2-pv1`) | lost unless the shard size moves | unchanged from `proving-v1.md` |
| NVIDIA 16 GB | yes | prove-only, miner paused per shard | kept | unchanged |
| NVIDIA 24 or 32 GB | yes | mines and proves (peak 16.8 GB on empty shards, 30.0 GB on a full prototype shard beside the miner) | kept | unchanged |
| Pool user | through the pool | the pool's own NVIDIA cards prove the shards assigned to the pool's keys (approximate: the pool protocol, spec 09, does not yet say who proves) | by the pool's rules | open, spec 09 |
## 4. The options
### 4a. CPU proving at a small shard size, as a tier
What it is: an AMD-only or Apple machine proves shards cut at a smaller budget than `S_p` on its CPU, through the
same host (`SP1_PROVER=cpu`; the host's `--budget` re-plan from branch `proving-v1`, commit c2544be, cuts a fixture at
any budget). The miner keeps the card; the prover takes the CPU.
What the numbers say: the fixed cost kills it. 282 s a shard on a 16-core PC and 355 s on the loaded M5 Max, with 30 GB of RAM, at the smallest shard there is; the time sits in the compressed-proof recursion, not in the cycles, so cutting shards smaller does not help, and the launch deadline (20 to 60 s behind the tip) is missed by 5x. It fits only the v1 unproven deadline (600 s), which pays a CPU prover only when no card has proven the shard in 10 minutes: on a chain with one NVIDIA prover that never happens. Recommendation: **no CPU tier**. Settings may still switch the CPU prover on (it does on macOS today), and the Proving tile must then say the proof takes about five minutes and is paid only when no card proves first.
What it costs the chain: a block cut into more, smaller shards costs more aggregation work (the aggregator guest
verifies one deferred proof per shard; 1.66 M cycles for four shards on the executor, bench-log 4 October; the
chained aggregation is 9.6 to 9.7 s per block on a mining 5090, `chain-pc2-pv1c`) and more records; the assignment
rule (8 assignees, 10 s window, spec 7.2) would need a CPU class with a longer window or the CPU provers only ever
win the open phase. None of that is measured. Status: Designed, nothing implemented.
### 4b. A second prover backend behind the swappable interface
The seam exists: `proving/igneum-prove/host/src/proof_system.rs` (`ProofSystem` trait, `Sp1ProofSystem`,
`StubProofSystem`), versioned per the design. The candidates:
| Target | Most likely backend | What exists | What adopting it costs |
|---|---|---|---|
| Apple silicon | RISC Zero's Metal prover (`metal` feature, shipped) | a shipped feature with kernels in the tree; ICICLE's Metal backend as the other library | a second guest program (the shard statement, `core/` is plain Rust and ports; the precompile patches for keccak and secp256k1 differ), a second pinned program id and verifying key in `elf/manifest.json`, the node's verifier for both proof formats (RISC Zero receipt and SP1 compressed proof) in `--mode verify` and the proof pool, and an aggregation problem: SP1's aggregator folds SP1 proofs by deferred verification; it cannot fold a RISC Zero receipt, so a block with shards from both families needs two aggregations or a wrapper. Approximate: weeks of a person's time, no measurement of a Metal shard time exists; RISC Zero's Groth16 wrapper for light clients does not run on Apple silicon at all |
| AMD | nothing | sppark's limited HIP path (not in SP1's tree); ICICLE's and Jolt's Vulkan and Metal work are not AMD | no backend to adopt. The honest statement is that it lands when a zkVM ships one |
### 4c. The public line
The site says today (read 5 October 2026 from `site/litepaper.html`, `site/miner.html`, `site/index.html`): "The same
card proves every block", "The card mines and proves", "Ember finds your GPU, makes a wallet for you and runs the
node, the miner and the prover as one app", "Target: shard size will be set so a 12 GB card proves one shard in about
20 seconds". Every one of those is true of an NVIDIA card with enough memory and false of an AMD or Apple card, and the
litepaper's own rule is "If consumer GPUs cannot prove shards fast enough, Igneum says so and does not launch on promises".
The recommended line, for the litepaper's proving section, the miner page and the app's Proving tile (copy law):
> Proving needs an NVIDIA card with 16 GB or more today (20 GB to mine and prove on the same card). AMD and Apple
> cards mine. A prover for them lands when a zkVM ships one. A CPU can prove a small shard in about five minutes with 32 GB of RAM free; the chain pays the first proof, which a card delivers in seconds, so CPU proving is for testing, not income.
Where the numbers come from: 16 GB and 20 GB are the measured gates of `proving-v1.md` (13.8 GB prover-alone peak,
16.8 GB mine-and-prove peak); "when a zkVM ships one" is section 1. The line changes when the memory sweep moves the
gates or a backend ships; it is reviewed with every prover release.
## 5. What this analysis does about it (the consequences, before anyone asks)
| Consequence | Action | Owner |
|---|---|---|
| An AMD-only miner loses the proving share | the CPU tier of 4a is measured here (section 2); whether it becomes a tier is a decision for Josh on those numbers | this analysis; Josh |
| The rig's prover unit must select NVIDIA cards only | already true in `prover_decision`; told the rig-installer agent to keep it as a stated rule and to print the CPU-fallback line for AMD-only rigs | rig-installer agent |
| The app's Proving tile on an AMD-only or Apple machine should say why it is off and name the CPU path | the `provedefault.rs` lines already say so for Apple; AMD-only Windows machines get "no NVIDIA card ..." | proving agent (told) |
| The site and litepaper over-promise for AMD and Apple | the line of 4c, to land with the next site pass (copy law; `node site/build.mjs`; link-check) | site-pages owner; not changed here |
| A Metal prover is the only non-NVIDIA path with a shipped backend | 4b names RISC Zero's Metal path and its cost; no work started | proving agent (told) |
## 6. Commands, jobs and sources
| What | Where |
|---|---|
| The PC 1 jobs (signed `run` jobs, PowerShell, not elevated, miners untouched, SP1_PROVER=cpu, host built without the `cuda` feature) | `tools/amd-prove/pc1-cpu-prove.ps1` (small fixtures), `pc1-cpu-prove-sp.ps1` (the `S_p` shard); published as `cpu-prove-pc1-small` (built, proved nothing: the quote bug) and `cpu-prove-pc1-small2` (the numbers) by `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --shell powershell --timeout-minutes 60`; `cpu-prove-pc1-sp` written, not published; `check-job-bash.sh` gates the bash body of every job script here; the package `igneum-prove-wsl2-pv1b.zip` (sha256 df50dee5...), the URL and hash filled at publish time, never committed |
| The Mac run | `tools/lock/with-lock.sh measure /usr/bin/time -l igneum-prove-host block-56-transfers-3shards.json --mode shard --shard 0` with the `proving-v1` worktree's host (pinned ids checked with `--mode id`) |
| Results | `node tools/jobs.mjs cpu-prove-pc1-small2 --all`, `docs/bench-log.md` entry "5 October 2026, the CPU prover on PC 1 and the backend survey" |
| Pages read | SP1: docs.succinct.xyz hardware-acceleration page, github.com/succinctlabs/sp1 (releases, `crates/sdk/src/lib.rs`, `sp1-gpu/README.md`, code search); RISC Zero: dev.risczero.com local-proving, `risc0/zkvm/Cargo.toml`; Jolt: README, book, PR #1733; OpenVM: releases; ICICLE: install_gpu_backend page, releases, README, PRs #735 and #1019; sppark README |

View file

@ -0,0 +1,418 @@
# ASIC resistance, 2011 to 2026: the history, the papers, the lessons, and the audit of Igneum against them
5 October 2026 (night), branch `asic-history`. Asked by Josh at 20:05 UTC: "do a full on deep dive into the full history of 'asic resistance' and see if we can add or upgrade anything." Baseline for the audit: the Counter ASIC 2.0 final class decided tonight (`docs/plans/counter-asic-2-status.md` on `ca2-coord`, entries 20:16 to 22:25 UTC; `docs/analysis/chip-model-v3.md` on `ca2-mixer` 1ab8b21). Every figure about another chain cites a repo file, a paper or a dated article, or is labelled approximate. Hash-per-joule gains are computed from the cited hashrate and watt figures of the chip and of the best consumer GPU of the same year, and are approximate by construction (GPU figures vary by tuning). Research gathered by four sub-agents between 20:10 and 20:45 UTC; the fetch failures they reported are listed in section 6.
## 0. One page for Josh
**What the history says Igneum is doing right.**
| # | What | The evidence |
|---|---|---|
| 1 | Binding the hash to random reads over a dataset larger than any on-chip cache, with the dataset growing on a schedule, and measuring the latency-bound share per card | Every compute-bound hash fell to a chip at 20x to 1,200x per joule within 16 to 37 months (rows Scrypt, X11, Blake, kHeavyHash, Blake3). The memory-bound hashes capped the chip at 1.1x to 4.8x (Ethash rows) or saw no chip at all (KawPow, Verthash, Autolykos, FishHash). RandomX's own design chose a 2 GiB dataset because a 2 GiB SRAM die was "questionable" at 7 nm in 2019 (RandomX `doc/design.md`) |
| 2 | A random program per epoch from a VDF seed, a weak-program filter, era draws from chain state, instruction families unlocked by height, and no scheduled human fork | Monero forked four times in 20 months and lost 85% of its hashrate at each fork to chips that returned within months (CryptoNight rows). Vertcoin forked three times and was 51%-attacked after two of them. Ravencoin's X16Rv2 fork was followed by FPGA bitstreams within weeks. Grin's six-monthly tweaks worked only because the lane was scheduled to die. The only random-program hashes with no chip after five years are the ones that never needed a fork (KawPow, FiroPoW, ProgPowZ) |
| 3 | Pricing the on-die-cache recompute chip and spending the mixer budget against it (x8: 0.92x with the 3x factor), with the model public | This is the "light-evaluation attack" Least Authority flagged on ProgPoW in 2019 and ProgPoW never fixed; Bob Rao's hardware audit put an on-die-DAG ProgPoW chip at "<< 0.1x" the energy per hash of a GPU. Kik's 2020 exploit was the same attack through a 64-bit seed. Igneum has a number against it; ProgPoW had a suggestion |
**What the history says Igneum is missing or under-weighting.**
| # | What | The evidence |
|---|---|---|
| 1 | The partial-store chip with a custom memory system (HBM or many narrow DRAM channels) is not in the chip model. The model prices only the f = 0 endpoint (all SRAM, recompute everything) | The only chip class that ever beat a memory-bound GPU hash did it this way: Ethash chips reached 2.1x (Linzhi, 2020), 2.9x (E9, 2022) and 4.8x (Jasminer X4, 2021) per joule through custom memory controllers and on-package memory, with no on-die dataset at all. The time-memory curve between f = 0 and f = 1 is open item O-1.6 and has never been drawn (`proto-metal/MEMHARD.md` section 3 item 2). Cuckoo Cycle's "tmto-hard" claim fell to a 50x memory cut for 2x time within two months of publication (Andersen, 2014) |
| 2 | The item-derivation mixer has a fixed shape. That fixed shape is exactly what hands the recompute chip its 3x fixed-function factor (bare 0.31x becomes 0.92x) | RandomX made the item derivation itself a random program (SuperscalarHash: about 450 instructions generated per seed, scheduled for a superscalar core, 170-cycle latency to match DRAM), so a chip cannot hard-wire it. Igneum draws the mixer's constants per day and keeps the shape; a chip hard-wires the shape. CryptoNight-R's random math per block raised chip latency only 2.5x; the lever is in the memory path, which for the recompute chip is the mixer |
| 3 | No clock and no detector. Chips have appeared at market caps from $18M (Radiant) and $23M (Grin) upward, and Vorick's 2018 rule was "any coin with over $20M of block reward in a year has a secret ASIC on it". Monero's secret chips held 85% of the hashrate before anyone saw them. Igneum's bounty is unfunded (D11) and the public benchmark is January 2027 | Rows Monero, Radiant, Grin, Handshake, Kadena; section 2.5 (the market-cap table); MoneroCrusher's nonce analysis (February 2019) found the chips by their share pattern, which is the detector Igneum can run from day one on the observer |
**The one change I would make first.** Add the partial-store HBM chip to the chip model and draw the time-memory curve before genesis (ranked addition 1, section 4.3). It is an analysis, it costs the GPU nothing, and it is the one chip class that history shows beating memory-bound GPU work. If that row comes out under 2x the x8 decision stands as measured; if it does not, the next lever is known before the vectors are frozen.
Everything else in this document is the evidence behind that page.
## 1. The history, one row per attempt
Columns: what the hash relied on; when it went live; the first chip that beat it (vendor, model, date, rated hashrate and watts); the gain in hash per joule against the best consumer GPU of the time (approximate, derived); how long it held (months from live to first chip); what the chain did. "None" in the chip column means no shipped chip was found in any source as of October 2026.
### 1.1 The rows
| # | Hash (chain) | Live | Relied on | First chip (vendor, model, date, rate, watts) | Gain per joule vs GPU (approximate) | Held (months) | Response | Sources |
|---|---|---|---|---|---|---|---|---|
| 1 | Scrypt (Tenebrix, Litecoin, Dogecoin) | Sep and Oct 2011 | 128 KB scratchpad, meant to fit a CPU cache and not a 2011 GPU; latency at SRAM scale | Gridseed GC3355, early 2014, 360 kH/s at 7 to 8 W; Innosilicon A2 Terminator, Apr 2014, 28 nm; KnC Titan, 2014, 300 MH/s at 850 W | Gridseed 19x, A2 68x, Titan 140x (vs Radeon 7970 at 700 kH/s, 285 W); Antminer L7 (2021) 1,100x | 27 | Embraced. Dogecoin merge-mined with Litecoin from Sep 2014 | [S1] [S2] [S3] [S4] |
| 2 | X11 (Dash) and the chain family X13, X15, X17, Quark, Nist5, C11 | Jan 2014 | A chain of 11 SHA-3 candidates; compute only. Duffield said it was meant to replay Bitcoin's CPU to GPU to ASIC path, not to prevent it | iBeLink DM384M, Mar 2016, 384 MH/s at 715 W; Baikal Giant A900 (2016); Antminer D3, Sep 2017, 19.3 GH/s at 1,200 W | DM384M 33x, D3 1,000x (vs R9 280X at 4 MH/s, 250 W) | 26 | Embraced. Baikal's multi-algo units covered the whole family by 2017 | [S5] [S6] [S7] |
| 3 | Ethash (Ethereum) | Jul 2015 | DAG of 1 GB growing per epoch, 128-byte random reads; memory bandwidth | Antminer E3, announced Apr 2018, 180 MH/s at 800 W, 4 GB DDR3; Innosilicon A10 Pro (2020) 500 MH/s at 950 W; Linzhi Phoenix (Dec 2020) 2,733 MH/s at about 3,000 W; Jasminer X4 (Oct 2021) 2.5 GH/s at 1,200 W; Antminer E9 (Jun 2022) 2.4 GH/s at 1,920 W | E3 1.1x to 1.6x (a tuned 1080 Ti beat it per watt); A10 Pro 1.2x vs RTX 3080; Phoenix 2.1x; X4 4.8x; E9 2.9x | 32 to the first chip; about 65 to a chip over 2x | ProgPoW (EIP-1057) debated 2018 to 2020 and shelved; PoS at the Merge, 15 Sep 2022. ASIC share of hashrate stayed small (one 2018 estimate: 3%) | [S8] [S9] [S10] [S11] [S12] [S13] |
| 4 | Etchash (Ethereum Classic) | Nov 2020 (Thanos, ECIP-1099) | Ethash with the DAG cut to 2.5 GB to keep 3 to 4 GB cards mining; not an anti-chip change | Post-Merge Ethash chips moved over: Antminer E9 Pro (Feb 2023) 3.68 GH/s at 2,200 W; Jasminer X16-P (2023) 5.8 GH/s at 1,900 W | E9 Pro about 4x, X16-P about 7x vs RTX 3090 (approximate) | n/a | Embraced | [S14] [S15] |
| 5 | Equihash 200,9 (Zcash, Horizen, Pirate) | Oct 2016 | Generalised birthday problem (Wagner); 144 MB in practice; memory size, with a claimed 1,000x compute penalty for halving memory | Antminer Z9 mini, announced 3 May 2018, 10 kSol/s at 300 W; Innosilicon A9 ZMaster (Jun 2018) 50 kSol/s at 620 W; Z11 (2019) 135 kSol/s at 1,418 W; Z15 (2020) 420 kSol/s at 1,510 W | Z9 mini 12x, A9 29x, Z11 34x, Z15 100x (vs GTX 1080 Ti at 700 Sol/s, 250 W) | 18 | Zcash: no fork (Zcon0 vote 45 to 19 against prioritising resistance, Jun 2018; ECC chose Sapling over resistance); PoS plan announced Nov 2021. Horizen: stayed after a 51% attack (Jun 2018). Pirate: stayed | [S16] [S17] [S18] [S19] [S20] |
| 6 | Equihash parameter forks: Zhash 144,5 (Bitcoin Gold), ZelHash 125,4 (Flux), BeamHash I to III 150,5 (Beam), 210,9 (Aion), 192,7 (Zero) | Jul 2018 (BTG), Jun 2019 (Flux), Jan 2019 (Beam) | Larger memory per solver than 200,9 (Beam and Flux also changed the datapath against FPGA bitstreams) | None found for 144,5, 125,4 or 150,5. Vorick (May 2018) wrote that an Equihash chip able to follow any parameter fork had been designed | n/a | BTG 7 years, Flux 7 years, Beam 7 years, with small prizes (section 2.5) | BTG: forked after the May 2018 51% attack, attacked again Jan 2020. Beam: planned "one or two hard forks" then BeamHash III (Jun 2020) as the last. Flux: stayed on ZelHash | [S21] [S22] [S23] [S24] |
| 7 | Scrypt-N, Lyra2RE, Lyra2REv2 (Vertcoin) | Jan 2014, Dec 2014, Aug 2015 | Memory-hard sponge (Lyra2) inside a hash chain; Scrypt-N grew N over time | Dayun Zig Z1, Sep 2018, 6.8 GH/s at 1,200 W (FPGA bitstreams of about 216 MH/s per board preceded it in 2018) | Z1 20x (vs GTX 1080 Ti at 59 MH/s, 207 W) | 37 (Lyra2REv2) | Lyra2REv3, Feb 2019, "to rid the network of the current generation of ASICs and FPGAs"; 51% attacks via rented hash in Oct to Dec 2018 (22 reorgs) and Dec 2019 | [S25] [S26] [S27] [S28] |
| 8 | Verthash (Vertcoin) | Jan 2021 | A 1.2 GB file generated from the chain's own block headers; random reads; bandwidth, like Ethash | None found | n/a | 69 and counting, small prize | No fork since | [S29] [S30] |
| 9 | X16R (Ravencoin) | Jan 2018 | 16 hashes in an order set by the previous block hash; compute, with order randomised | OW Miner OW1, Sep 2019, about 182 MH/s at 1,400 W; SKC Turing R1 claimed. Widely thought FPGA-based | OW1 1.3x (vs GTX 1080 Ti at 18 to 31 MH/s, 190 to 284 W) | 20 | X16Rv2, 1 Oct 2019 (hashrate fell 70%); FPGA bitstreams for X16Rv2 within weeks (BittWare CVP-13 at 240 MH/s) | [S31] [S32] [S33] |
| 10 | KawPow (Ravencoin; Neoxa, Clore, Meowcoin, Neurai) | 6 May 2020 | ProgPoW 0.9.4 variant: random math per block, DAG, 16 KB cache reads; targets the GPU datapath | None found as of 2026 (no vendor lists one; one retailer listing naming an "Antminer X9" for KawPow is an error) | n/a | 77 and counting | Roadmap: "No additional future algorithm forks are envisaged" | [S34] [S35] [S36] |
| 11 | MTP (Zcoin, now Firo) | Dec 2018 | Argon2d memory array with a Merkle tree (Biryukov and Khovratovich, "Egalitarian computing"); 4 GB; memory size | None | n/a | 34, then replaced | Dinur and Nadler broke the 2 GB instance to under 1 MB at a 170x compute penalty before launch (2017); MTP 1.2 patched it. FiroPoW (ProgPoW variant) Oct 2021 for block size and GPU fairness, not for a chip | [S37] [S38] [S39] [S40] |
| 12 | FiroPoW (Firo) | 26 Oct 2021 | ProgPoW 0.9.4 with a per-block program | None found | n/a | 59 and counting | Nov 2025 fork cut the maximum DAG to 6.76 GB to keep 8 GB cards | [S40] [S41] |
| 13 | Blake-256 14r (Decred) | Feb 2016 | Compute; chosen to be ASIC-friendly ("easy and fast implementation of hardware is the main design goal") | Innosilicon D9, about Apr 2018, 2.4 TH/s at 1,000 W; Obelisk DCR1 (Jun 2018); Antminer DR5 (Dec 2018) 35 TH/s at 1,610 W | D9 130x, DR5 1,200x (vs GTX 1080 Ti at 4.6 GH/s, 250 W) | 26 | Embraced by design | [S42] [S43] [S44] |
| 14 | Blake2b (Sia) | Jun 2015 | Compute | Antminer A3, Jan 2018, 815 GH/s at 1,186 W; Obelisk SC1 (Jul 2018) 550 GH/s at 500 W; Innosilicon S11 (2018) 3.83 TH/s at 1,380 W | A3 58x, S11 230x (vs GTX 1080 Ti at 2.96 GH/s, 250 W) | 31 | Fork at block 179,000 (Oct 2018) to brick Bitmain and Innosilicon units and keep Obelisk's; Innosilicon then held about 37% of hashrate | [S45] [S46] [S47] |
| 15 | Blake2s (Kadena) | Nov 2019 | Compute; chosen to be "GPU mineable and not immediately ASIC mineable (but for which an ASIC can be made)" | Goldshell KD2 and KD5, Mar 2021, 18 TH/s at 2,250 W; Antminer KA3 (Sep 2022) 166 TH/s at 3,154 W | KD5 195x, KA3 1,280x (vs RTX 3080 at 8.9 GH/s, 217 W) | 16 | Embraced by design | [S48] [S49] [S50] |
| 16 | CryptoNight (Bytecoin, Monero) | Jul 2012, Apr 2014 | 2 MB scratchpad sized to a per-core L3, AES rounds, random reads; latency at SRAM scale | Secret chips from about late 2017 (85% of the hashrate vanished at the April 2018 fork); Antminer X3, announced Mar 2018, 220 kH/s at 550 W; Baikal Giant-N | X3 40x to 50x (vs Vega 64 at about 2 kH/s, 200 to 250 W, approximate) | 43 to the secret chips, 47 to the announced one | Forks: CryptoNight v7 (6 Apr 2018), v8 (18 Oct 2018), CryptoNight-R (9 Mar 2019, random math per block seeded by height, chip latency up 2.5x), RandomX (30 Nov 2019). MoneroCrusher's nonce analysis (Feb 2019) found chips at over 85% of the hashrate again, four months after v8 | [S51] [S52] [S53] [S54] [S55] |
| 17 | RandomX (Monero; Wownero, ArQmA, Zephyr, Tari) | 30 Nov 2019 | A VM running 8 chained random programs per hash on a superscalar CPU with floating point; 2 GiB dataset derived from a 256 MiB cache by a random superscalar program (SuperscalarHash); 2 MiB scratchpad in L1, L2, L3 tiers | Antminer X5, Sep 2023, 212 kH/s at 1,350 W (RISC-V cores); Antminer X9, Jul 2026 delivery, 1 MH/s at 2,472 W; Pinecone INIBOX R1X, Mar 2026, 1.2 MH/s at 2,055 W | X5 at parity with a Ryzen 9 7950X (157 against about 200 H/J, approximate); X9 and R1X 2x to 3x over the best CPU (approximate). GPUs are 25x worse per joule than CPUs on it | 46 to parity hardware, about 75 to a 2x to 3x chip | No fork as of Oct 2026. Four audits in 2019 (Trail of Bits, X41, Kudelski, QuarksLab) found nothing critical | [S56] [S57] [S58] [S59] [S60] |
| 18 | Cuckoo Cycle (Grin Cuckaroo lane, Aeternity, Cortex) | Jan 2019 (Grin) | Find a 42-cycle in a random graph; lean solver one bit per edge; memory latency, or bandwidth in the mean solver | None on the Cuckaroo lane (Cuckaroo29 tweaked every 6 months: Cuckarood Jul 2019, Cuckaroom Jan 2020, Cuckarooz Jul 2020) | n/a | 24, retired on schedule | The lane was built to die: 90% of reward at launch falling to 0% in Jan 2021 (HF4) | [S61] [S62] [S63] |
| 19 | Cuckatoo31+ (Grin's chip lane) | Jan 2019 | Same, with plain bits in place of ternary counters to simplify chips; "Proof of SRAM" per Tromp | Obelisk GRN1 announced Jan 2019 and cancelled Jul 2019; Innosilicon G32 announced 2019 and never shipped; iPollo G1, Dec 2020, 36 GPS Cuckatoo32 at 2,800 W | G1 about 4x (vs RTX 3090 at about 1 GPS, 300 W, approximate) | 23, by design | Surrender by schedule. Tromp's $10,000 linear TMTO bounty was claimed in Apr 2025 (N/k bits at about k + 1,000 hashes per edge) | [S61] [S64] [S65] [S66] |
| 20 | ProgPoW (Ethereum proposal; Bitcoin Interest, Sero, Zano as ProgPowZ, Quai) | EIP May 2018; Bitcoin Interest 2018; Quai Jan 2025 | Random math per period on a 32-register file, 16 KB cache reads, 256-byte DAG loads, keccak-f800; "saturate the GPU" | None | n/a | 8 years across its adopters | Ethereum: tentatively approved Jan 2019 and Feb 2020, petition 27 Feb 2020, left "approved" and unscheduled on 6 Mar 2020, dead. Audits: Least Authority (Sep 2019) and Bob Rao (Sep 2019). Kik's 64-bit-seed exploit (Mar 2020) patched in 0.9.4 | [S67] [S68] [S69] [S70] [S71] [S72] |
| 21 | Autolykos v1 and v2 (Ergo) | Jul 2019; v2 Feb 2021 | v1: memory-hard with a per-miner secret key, so puzzles could not be outsourced to pools. v2: the secret removed (contract pools bypassed it); a 2 GB table that grows 5% per 51,200 blocks from block 614,400 | None | n/a | 87 and counting, small prize | v2 by EIP-0009 at block 417,792 | [S73] [S74] |
| 22 | Octopus (Conflux) | Oct 2020 | Ethash-style DAG; the "dense matrix step" could not be verified in `conflux-rust` tonight (unverified) | None | n/a | 72 and counting | CIP-102 (Aug 2022) proposed switching to Ethash to attract post-Merge miners; dormant | [S75] [S76] |
| 23 | kHeavyHash (Kaspa; Bugna kept it) | Nov 2021 | cSHAKE256, a 64x64 4-bit matrix multiply from the pre-PoW hash, cSHAKE256; compute, designed for optical and specialised hardware | IceRiver KS0, Jul 2023, 100 GH/s at 65 W; KS1, KS2 (Sep 2023); Antminer KS3, Aug 2023, 8.3 TH/s at 3,188 W; KS5 Pro (Mar 2024) 21 TH/s at 3,150 W | KS0 250x, KS5 Pro 1,100x (vs RTX 3090 at 910 MH/s, 150 W) | 17 to 20 | Embraced (Sompolinsky, May 2023: "an overall positive"). Hashrate went from under 100 PH/s to over 700 PH/s in months; the GPU share was negligible by late 2023 (approximate). Forks that left: Karlsen (FishHashPlus, Sep 2024), Pyrin (PyrinHash v2, Sep 2024), Spectre (CPU AstroBWTv3), Nexellia, Waglayla, Cryptix, Hoosat | [S77] [S78] [S79] [S80] [S81] |
| 24 | NexaPow (Nexa) | 2023 | SHA-256 plus a secp256k1 Schnorr signature per attempt; framed as "useful ASICs" | DragonBall A21, Jan 2025, 3.4 GH/s at 1,800 W | 3x to 4x (vs RTX 3090 at 123 to 137 MH/s, 230 W, approximate) | 24 | None | [S82] [S83] |
| 25 | Blake3 (Alephium; Iron Fish until 2024) | Nov 2021 | Double Blake3; compute; chosen as ASIC-friendly | Goldshell AL-BOX, 2023, 360 GH/s at 180 W; IceRiver AL0; Antminer AL1 (2024) 15.6 TH/s at 3,510 W; AL3 | AL-BOX 156x, AL1 350x (vs RTX 3090 at 2.3 GH/s, 180 W) | 22 to 24 | Alephium embraced. Iron Fish forked to FishHash (Apr 2024, FIP-3: Ethash-derived, fixed 4.6 GB dataset, 512 iterations, 128-byte mix); Karlsen adopted FishHashPlus (Sep 2024) | [S84] [S85] [S86] [S87] |
| 26 | FishHash (Iron Fish, Karlsen) | Apr 2024 | Ethash-derived, 4.6 GB fixed dataset; bandwidth | None found | n/a | 30 and counting, small prize | None | [S87] |
| 27 | Eaglesong (Nervos) | Nov 2019 | Compute (a new sponge) | Toddminer C1 (Feb 2020); Antminer K5, Mar 2020, 1.13 TH/s at 1,580 W; Goldshell CK5 (Mar 2021) 12 TH/s at 2,400 W | K5 70x, CK5 500x (vs RTX 3090 at about 2.1 GH/s, approximate) | 4 | Embraced | [S88] [S89] |
| 28 | Blake2b + SHA3 (Handshake) | Feb 2020 | Compute | Goldshell HS1, Jun 2020; HS3 (Jul 2020) 2 TH/s at 2,000 W; HS5 | Over 100x (approximate) | 5 | Embraced | [S90] [S91] |
| 29 | SHA512/256d (Radiant) | 2022 | Compute | DragonBall A11; IceRiver RX0, Sep 2024, 260 GH/s at 100 W | About 550x (vs RTX 3090 at 1.3 to 1.5 GH/s, approximate) | About 24 | None | [S92] [S93] |
| 30 | ProgPowZ (Zano), DynexSolve (Dynex), Janushash (Warthog), XelisHash v1 and v2 (Xelis), VerusHash 2.2 (Verus) | 2019 to 2024 | ProgPoW variant; GPU "neuromorphic" useful work; a product of VerusHash and SHA256t to balance CPU and GPU; CPU and GPU balanced; AES-based CPU hash | None found for any of them | n/a | Small prizes throughout | Xelis forked to v2 (Jul 2024) for FPGA resistance | [S94] [S95] [S96] [S97] [S98] |
| 31 | Ethash on EthereumPoW (ETHW) after the Merge | Sep 2022 | As Ethash | The Ethash chips above | About 4x (E9 Pro, X16-P vs RTX 3090, approximate) | n/a | Embraced | [S15] |
### 1.2 What the rows say when sorted
| Class of hash | Rows | Months to first chip | First-chip gain per joule | Best gain reached |
|---|---|---|---|---|
| Compute only (chains of hashes, Blake family, SHA-3 family, matrix multiply) | 2, 13, 14, 15, 23, 25, 27, 28, 29 | 4 to 31 (median about 24) | 33x to 250x | 500x to 1,280x |
| Memory at SRAM scale (128 KB scrypt, 2 MB CryptoNight) | 1, 16 | 27, 43 | 19x, 40x to 50x | 1,100x (Scrypt, 2021) |
| Memory size without a bandwidth bound (Equihash, Lyra2REv2) | 5, 7 | 18, 37 | 12x, 20x | 100x |
| Memory bandwidth at DRAM scale (Ethash, Verthash, FishHash, Etchash) | 3, 4, 8, 26 | 32 (Ethash); none for the others | 1.1x to 1.6x | 2.9x to 4.8x |
| Random program on a commodity datapath (RandomX, ProgPoW family, X16R's order randomisation) | 9, 10, 12, 17, 20, 30 | X16R 20 (FPGA-class, 1.3x); RandomX 46 to parity; none for ProgPoW's adopters in 8 years | 1.3x (X16R), 1x (RandomX 2023) | 2x to 3x (RandomX 2026, approximate) |
| Graph search (Cuckoo) | 18, 19 | 23 on the chip lane; never on the tweaked lane | 4x | 4x |
Two caveats on the random-program rows. The prizes were small: Ravencoin, Firo and Zano never reached the market caps at which the 2018 chips appeared (section 2.5), so "no chip" is partly an economic fact. And RandomX's chips arrived once Monero's reward justified them: parity hardware at 46 months, a 2x to 3x chip at about 75 months (approximate), on a hash whose whole purpose was to make the CPU the chip.
## 2. The academic side
### 2.1 Memory-hard functions
| Paper | Result | What it means for Igneum |
|---|---|---|
| Abadi, Burrows, Manasse, Wobber, "Moderately hard, memory-bound functions", NDSS 2003 and ACM TOIT 2005 [P1]; Dwork, Goldberg, Naor, "On memory-bound functions for fighting spam", CRYPTO 2003 [P2] | The origin of the idea: CPU speed varies 100x across machines, memory latency does not, so a cost function bound by cache misses is fairer than one bound by cycles | Igneum's latency-bound rule is this argument from 2003 applied to GPUs and DRAM: the DRAM row cycle is the same physics for a chip and a card (section 2.6) |
| Percival, "Stronger key derivation via sequential memory-hard functions", BSDCan 2009 [P3] | Defines sequential memory-hardness; ROMix is sequential memory-hard in the random-oracle model; cost measured in area-time (dollar-seconds) | The area-time measure is the one the chip model uses (equal silicon); scrypt's 2011 deployment at 128 KB ignored the paper's own scale |
| Alwen and Serbinenko, "High parallel complexity graphs and memory-hard functions", STOC 2015 [P4] | Cumulative memory complexity (CMC) in the parallel random-oracle model; earlier sequential measures fail against parallel, amortising adversaries | A chip is a parallel, amortising adversary; any Igneum claim about the dataset must be made in a parallel model |
| Alwen and Blocki, "Efficiently computing data-independent memory-hard functions", CRYPTO 2016 [P5]; "Towards practical attacks on Argon2i and Balloon hashing", EuroS&P 2017 [P6] | Any data-independent MHF can be computed in less than n^2 cumulative memory; Argon2i at O(n^1.75 log n), Catena and Balloon at O(n^1.67); the attacks are practical at real parameters | Igneum's addresses are data-dependent (register state), which is the right side of this result; the price is cache-timing leakage, which does not matter for a PoW |
| Alwen, Chen, Pietrzak, Reyzin, Tessaro, "Scrypt is maximally memory-hard", EUROCRYPT 2017 [P7] | scrypt's CMC is Omega(n^2 w) in the parallel ROM, optimal, against parallel amortising adversaries | Data-dependent chains of reads are the construction with the proof; Igneum's item derivation (8 dependent cache reads) is a short chain of this kind, with no proof |
| Biryukov, Dinu, Khovratovich, "Argon2", EuroS&P 2016 [P8]; Boneh, Corrigan-Gibbs, Schechter, "Balloon hashing", ASIACRYPT 2016 [P9] | Argon2d: a one-pass adversary can cut memory at most 3x at equal area-time; Argon2i needs over 10 passes to resist the Alwen-Blocki attack. Balloon: provable in the sequential model only; the paper says parallel ASIC attacks are outside its model | A "memory-hard" label without a stated adversary model has been wrong three times in this list (Argon2i, Balloon, Catena) |
| Biryukov and Khovratovich, "Tradeoff cryptanalysis of memory-hard functions", ASIACRYPT 2015 [P10]; Forler, Lucks, Wenzel, "Catena", 2013 [P11]; Simplicio et al., "Lyra2", IEEE TC 2016 [P12] | The ranking trade-off attack on Lyra2, yescrypt and Argon2; Catena's proofs flawed, 25x area-time cut; designers changed their algorithms | Lyra2REv2 (row 7) carried this construction into a PoW and still fell to a chip at 20x; the cryptanalysis found the shortcut before the chip did |
### 2.2 Bandwidth-hard functions
| Paper | Result | What it means for Igneum |
|---|---|---|
| Ren and Devadas, "Bandwidth hard functions for ASIC resistance", TCC 2017 [P13] | Memory-hardness (CMC) bounds a chip's area advantage and says nothing about energy; energy spent on off-chip memory traffic is comparable for a chip and a CPU, so bandwidth-hardness is the lever; scrypt, Catena-BRG and Balloon are bandwidth-hard with suitable parameters; the stacked double butterfly is capacity-hard and not bandwidth-hard | The chip model's "equal silicon" row is an area argument. The energy argument is the one the Ethash chips answered: they moved the same bytes at lower energy per byte with custom memory controllers (rows 3 and 4). Igneum's hash moves 128 x 64 B = 8 KB of DRAM lines per hash on AMD and 128 x 32 B on NVIDIA; a chip with 4-byte access granularity moves 512 B for the same work. That is the bandwidth-per-watt gain the plan's last section warns about, stated in Ren-Devadas's units |
| Blocki, Ren, Zhou, "Bandwidth-hard functions: reductions and lower bounds", CCS 2018 [P14] | Bandwidth cost in the parallel ROM equals the red-blue pebbling cost of the graph; high CMC implies high bandwidth cost; Argon2i and DRSample are maximally bandwidth-hard; a tight lower bound on scrypt's energy | The right formal target for a future proof about the item derivation, if one is ever attempted; none exists today |
| Alwen, Blocki, Harsha, "Practical graphs for optimal side-channel resistant MHFs", CCS 2017 [P15] | DRSample: a practical graph with maximal depth-robustness | Not applicable: Igneum does not need side-channel resistance |
### 2.3 Asymmetric, egalitarian and graph proofs of work
| Paper | Result | What it means for Igneum |
|---|---|---|
| Biryukov and Khovratovich, "Equihash", NDSS 2016 [P16] | Wagner's generalised birthday with algorithm binding; claimed 1,000x compute for halving memory | The claim did not survive contact with a chip design: 144 MB in practice fitted the Z9's memory system (row 5) |
| Biryukov and Khovratovich, "Egalitarian computing", USENIX Security 2016 [P17]; Dinur and Nadler, "Time-memory tradeoff attacks on the MTP proof-of-work scheme", CRYPTO 2017 [P18] | MTP: Argon2d plus a Merkle tree. Dinur-Nadler: malicious proofs with under 1 MB in place of 2 GB at a 170x compute penalty, by injecting blocks that steer Argon2d's data-dependent addressing | The attacker who controls the memory's contents controls the addresses. In Igneum the day key comes from a VDF of chain state and the cache fill is a chained block function, so no miner chooses the contents. The analogy still holds for the unreviewed mixer: a structural weakness in M_r is the shortcut this paper found in MTP |
| Tromp, "Cuckoo Cycle", BITCOIN 2015 [P19]; Andersen, "A public review of Cuckoo Cycle", 31 Mar 2014, and "Exploiting time-memory tradeoffs in Cuckoo Cycle", 1 Aug 2014 [P20]; the linear TMTO bounty, claimed Apr 2025 [S66] | Edge trimming cut memory about 50x for about 2x time, two months after publication; Tromp adopted it. The 2025 bounty result: an N/k-bit chip must hash each edge about k + 1,000 times | A time-memory claim is a curve, and the curve was wrong by 50x until someone drew it. Igneum's curve between "store everything" and "recompute everything" has not been drawn (O-1.6) |
| Georghiades, Flolid, Vishwanath, "HashCore", 2019 [P21] | "Inverted benchmarking": random widgets modelled on SPEC CPU workloads so the CPU is already the chip | The same idea as RandomX and ProgPoW stated generally: the hash is a benchmark of the target hardware |
### 2.4 Program-based proofs of work and their audits
| Document | What it says | What it means for Igneum |
|---|---|---|
| RandomX `doc/design.md` and `doc/specs.md` (tevador) [S56] [S57] | A VM so that the work is "data and code"; 8 chained programs per hash so a miner cannot filter (filtering 25% of programs at a 50% speedup yields 0.44x honest speed); SuperscalarHash of about 450 instructions with 155 multiplies, scheduled for a superscalar core at a 170-cycle latency to match DRAM, so a light-mode chip with the 256 MiB cache on die pays 760 cycles and 1,240 multiplies per item, "energy comparable to loading 64 bytes from DRAM"; a 2,080 MiB dataset because a 2 GiB SRAM die was "questionable" at 7 nm in 2019; cache-to-dataset ratio capped at 8 to keep the area-time product constant; double-precision floating point to force the whole CPU; "DRAM cannot do more than about 25 million random accesses per second per bank group" | Igneum rebuilt the idea for a GPU. The parts that carried over: the dataset above on-chip cache, the dependent item derivation, the cache-to-dataset ratio (4 at genesis, 8 at year 4 under option C). The parts that did not: per-hash programs (a GPU cannot JIT per hash and stay a GPU), floating point (vendor rounding), a random item-derivation program (Igneum's mixer has a fixed shape with drawn constants). Section 4.3 ranks the last of these |
| Trail of Bits audit of RandomX, 2 Jul 2019 [S60]; Kudelski, X41, QuarksLab (2019) | Two low findings and 47 brittle parameters; the design affirmed | Four paid external reviews before launch, for a hash whose whole value was the resistance claim. Igneum has had none (ledger M7) |
| EIP-1057 ProgPoW [S67]; Least Authority audit, 9 Sep 2019 [S69]; Bob Rao hardware audit, Sep 2019 [S70] | Claimed chip gain 1.1x to 1.2x. Least Authority: no issues, five suggestions, one of them the light-evaluation attack (on-the-fly DAG generation with the 16 MB cache in on-die SRAM) "may become possible within a few years" once about 100 MB of fast on-die SRAM is feasible. Rao: energy per hash is the only meaningful metric; shipping Ethash chips show about 1.6x hashrate per watt; conventional compute chips gain little on ProgPoW; integrating the DAG on die cuts data-movement energy by over 10x, so "ProgPOW ASICs with << 0.1X E/H over GPUs can be built"; an advanced-node chip is "$20M+" and "1+ year"; a 16-die split holding a 2.78 GB DAG was about $172 per board in 2019 against about $240 for a GPU board, and a monolithic die "viable around 2025" | The light-evaluation attack is Igneum's M16 recompute chip. ProgPoW left it as a suggestion; Igneum priced it and spent the mixer against it (x8). Rao's "$172 per board for a 16-die split" is the HBM-class partial-store chip in a different form, and it is the row the Igneum model lacks |
| Kik, "ProgPoW exploit", 4 Mar 2020 [S71] | A 64-bit seed lets a chip skip memory access with a cooperating node; patched in 0.9.4 | Igneum's seed is 256 bits and the program is the epoch's; the nearest analogue is header grinding for cache locality, unmeasured (section 4.3, check 4) |
### 2.5 The economics of a chip
**What a chip costs to make** (design plus masks, by node; all figures from the cited articles, which disagree with each other by 2x and say so):
| Node | Mask set | Full design, IBS as quoted by Semiengineering (2018 and 2021) | Full design, other estimates | Sources |
|---|---|---|---|---|
| 65 nm MPW shuttle | n/a | n/a | Europractice 2025: about €51,000 minimum (9 mm^2 at €5,720 per mm^2) | [E1] |
| 28 nm | "beyond $1M" (SemiAnalysis 2022); $1M to $3M (Silicon Analysts 2026) | $51.3M (2018); $40M (2021) | $5M to $30M total NRE for a small chip (Silicon Analysts) | [E2] [E3] [E4] |
| 16/12 nm | n/a | $106M (2018 revision of a 2014 $310M figure) | Europractice MPW 16 nm: about €125,000 minimum | [E1] [E4] |
| 7 nm | "beyond $10M" (SemiAnalysis); $5M to $10M (Silicon Analysts) | $297.8M (2018); $217M "mainstream" (2021); Semiengineering's own 2023 discount: about $160M | Startups shipped 7 nm chips for "$50M to $75M" all-in (SemiAnalysis); a 10 nm-class mining chip "$20M+" (Rao 2019) | [E2] [E3] [E4] [S70] |
| 5 nm | $10M to $20M | $542.2M (2018); $416M (2021); about $280M discounted (2023) | Marvell 2023: $449M (secondary source, approximate) | [E2] [E3] [E5] |
| 3 nm | "$40M range" | $500M to $1.5B (2018); $590M (2021) | Marvell 2023: $581M (secondary, approximate) | [E2] [E3] [E5] |
Miner-makers' own numbers: Bitmain's 2018 filing shows R&D of $73M in 2017 and $86M in the first half of 2018 and three failed chips at a reported combined cost of about $500M [E6]; Canaan's 2019 prospectus shows R&D of $26.5M in 2018 and "seven tape-outs" at a 100% success rate [E7]; Vorick wrote that Bitmain brought the Sia A3 to market for "less than $10 million" and took over $20M of orders within eight minutes [S47]; Obelisk's DCR1 was a 28 nm part [S43]; Taylor's 2013 survey gives $150,000 for a 130 nm and $500,000 for a 65 nm Bitcoin chip NRE in 2012 [E8].
**Where the 256 MiB cache lands a chip.** The Counter ASIC 2.0 analysis priced the 256 MiB SRAM mirror at 128 mm^2 and $46 per good die at N5 on the shipped-product density (`sram-mirror.md` revision 2, from AMD V-Cache 64 MB on 41 mm^2 at N7 [E9], TSMC N5 HD macro 31.8 Mib/mm^2 [E10]). The history adds the node question: a cheap chip is a 28 nm chip ($1M to $3M of masks, a $5M to $30M project), and a 28 nm bit cell is about 6x an N7 cell (approximate, from memory: TSMC 28 nm HD about 0.127 um^2 against N7's 0.027 [E10]), so 256 MiB at 28 nm is about 1,000 mm^2 of SRAM on the V-Cache density: more than a reticle. The cache forces the recompute chip onto a 7 nm or better node, which moves its project from the $5M class to the $50M class (SemiAnalysis's 7 nm startup figure). That is a stronger statement than the $46 per die, and it is the reason the cache size matters more than its per-die cost. Option C (the cache doubles with the dataset) keeps it true as nodes shrink: at the 6% per year density trend the status file cites, a 512 MiB mirror in year 4 costs more mm^2 than 256 MiB today.
**When chips appeared** (CoinMarketCap historical snapshots pulled by the research agent; daily issuance is arithmetic from each chain's schedule; all approximate):
| Chain | First public chip | Market cap then | Daily issuance then (USD) |
|---|---|---|---|
| Litecoin | Gridseed, Dec 2013 | $817M | $1.0M |
| Dash | PinIdea DR-100, Aug 2017 (iBeLink 2016 widely cited, date unverified) | $2.2B | $0.6M |
| Siacoin | Obelisk SC1 announced Jun 2017; Antminer A3 Jan 2018 | $430M; $1.5B | $0.43M; $1.1M |
| Decred | Obelisk DCR1 Jun 2017; Innosilicon D9 Apr 2018 | $216M; $353M | $0.18M |
| Monero | Antminer X3, Mar 2018 (secret chips from early 2017 per Vorick, unverified) | $3.3B | $0.75M |
| Ethereum | Antminer E3, Apr 2018 | $37.4B | $7.6M |
| Zcash | Z9 mini, May 2018 | $1.1B | $2.1M |
| Bitcoin Gold | the same chips, May 2018 | $1.3B | $0.14M |
| Grin | GRN1 announced Jan 2019 (cancelled); G32 Apr 2019 (never shipped); iPollo G1 Dec 2020 | $23M (Apr 2019); $23M (Dec 2020) | $0.24M; $33K |
| Nervos | Toddminer C1, Feb 2020 | $75M | $65K to $85K |
| Handshake | Goldshell HS1, Jun 2020 | $30M | $31K |
| Kadena | Goldshell KD5, Mar 2021 | $42M | $21K |
| Kaspa | IceRiver KS0, Jul 2023 | $480M | $0.42M |
| Alephium | Goldshell AL-BOX, May 2024 | $179M | $0.1M |
| Radiant | IceRiver RX0, Sep 2024 | $18M | $22K |
Sources: [E11] (the research agent's CoinMarketCap pulls, dates in the table) and the chip rows above. Reading: a compute-bound hash gets a chip at $20K to $30K of daily issuance (Radiant, Kadena, Handshake); the 2018 cluster sat at $0.15M to $2M a day. Vorick's rule from May 2018: any coin with over $20M of block reward in a year (about $55K a day) should assume a secret chip [S47]. For Igneum the clock is the day its issuance in dollars crosses about $50K; a memory-bound hash buys time against that clock (Ethash: 32 months at the largest prize in the table), and the random program buys more (section 1.2), but nothing in the table says it buys forever.
### 2.6 Latency as the resource
| Source | What it says | What it means for Igneum |
|---|---|---|
| Li, Reddy, Jacob, "A performance and power comparison of modern high-speed DRAM architectures", MEMSYS 2018 [L1] | Row timings from datasheets: DDR4 tRCD 14, tRAS 33, tRP 14 ns; GDDR5 tRCD 14, tRAS 28, tRP 12; HBM and HBM2 tRCD 14, tRAS 34, tRP 14. Row cycle tRC about 40 ns (GDDR5) to 48 ns (DDR4, HBM2). "The memory-latency problem does still remain" | The row cycle is the floor under every dependent random read whatever the controller; HBM does not shorten it. What HBM and a custom controller change is the number of rows that can be opened per second per watt (channels, banks, pseudo-channels), which is the throughput of random reads in flight, which is what the 9070 XT probe measured as the card's ceiling (2.4 G reads/s against the 5090's 17.5 G) |
| Chang, CMU thesis, Dec 2017 [L2] | Over two decades DRAM capacity improved 128x, bandwidth 20x, latency 1.3x | The latency-bound rule has a long half-life; the bandwidth-per-watt lever (the Ethash chips) does not stand still |
| NVIDIA profiling guide and the Ampere tuning deck [L3] | L1 and L2 lines are 128 bytes in four 32-byte sectors; DRAM-to-L2 transactions default to 64 bytes since Volta, configurable 32, 64 or 128 on A100 | The 5090's 32-byte sector per 4-byte read is where a custom controller gains bandwidth efficiency (8x fewer bytes), the Ren-Devadas energy lever; the 9070 XT's 64-byte line is 16x. Neither changes the row cycle |
| RandomX `doc/design.md` [S56] | About 25 million random accesses per second per DRAM bank group; "all Dataset accesses read one CPU cache line (64 bytes) and are fully prefetched"; one program iteration tuned to "typical DRAM access latency (50-100 ns)" | The same arithmetic Igneum uses (128 dependent reads per hash against the card's random-read ceiling), from the design that held longest |
| Condrey, "PoSME", arXiv Apr 2026 (single author, not peer reviewed) [L4] | Latency-bound pointer chasing with hash compute under 3.5% of the step cost; GPUs 14x to 19x slower than a consumer CPU | The only dedicated latency-bound PoW paper found; its GPU-vs-CPU gap is the cost RandomX pays, and the cost Igneum avoids by keeping thousands of loads in flight per card |
The paper that does not exist: nothing found treats cache timing as a feature; Catena and Argon2i treat it as a leak. No peer-reviewed survey of ASIC resistance as such surfaced; the nearest are Cho's 2018 multi-hash evaluation [P22] (X11-style resistance "is not strong enough"), Feng and Luo's 2020 three-processor study [P23] (GPUs dominate CryptoNight, Ethash and Cuckoo on CPU, GPU and Xeon Phi) and Yaish and Zohar's 2023 pricing of mining hardware as a bundle of options [P24].
## 3. The lessons
Each lesson is stated once, with the rows it comes from.
1. **Compute-bound work loses by 30x to 1,000x within two years, whatever its shape.** Chains of eleven hashes (row 2), sixteen hashes in a random order (row 9), a 64x64 matrix multiply (row 23), a new sponge (row 27), a signature per attempt (row 24): every one got a chip, the random-order chain at 1.3x by an FPGA within 20 months and the rest at 33x to 1,100x. Multiplying the number of fixed functions multiplies the chip's die, not its difficulty. For Igneum: nothing in the program's ALU work is a defence and the design already says so (ledger M1); the defence is the memory path.
2. **Memory at SRAM scale is compute-bound with extra steps.** Scrypt's 128 KB (row 1) and CryptoNight's 2 MB (row 16) were sized to a 2011 and a 2014 CPU cache; a chip put the same memory on die and won 19x and 40x. Igneum's answer is the 256 MiB cache growing with the dataset (option C) and the 1 GiB to 2 GiB dataset; section 2.5 shows the cache size also sets the chip's node and therefore its project cost. The hot table (layer 5) was a step back toward SRAM scale, and the measurement agreed (the honest card paid 7% to 16%, the chip paid $0.23 per MB).
3. **Bandwidth-bound work gets a memory chip at 2x to 5x.** Ethash held 32 months and then got chips whose whole design was the memory system: DDR3 (E3, no gain), GDDR6 (A10 Pro, 1.2x), custom controllers (Linzhi, 2.1x), on-package memory (Jasminer X4, 4.8x) (rows 3, 4). Rao's audit explains why in energy terms: the chip moves the same bytes at lower energy per byte, and a split-die design holding the DAG was already cheaper than a GPU board in 2019. Ren and Devadas give the bound: a chip's energy advantage on a bandwidth-hard function is the ratio of its memory energy per bit to the GPU's. Igneum's rule "avoid leaning on bandwidth" is right; its model has no row for this chip (section 4.3, addition 1).
4. **Latency-bound and random-program work held longest, and the prize was usually small.** CryptoNight's latency bound at SRAM scale held 43 months, then fell to secret chips (row 16). RandomX's at DRAM scale held 46 months to parity hardware and about 75 to a 2x to 3x chip (row 17, approximate), on the largest prize any resistant hash has carried. ProgPoW's adopters have had no chip in 8 years on small prizes (rows 10, 12, 20). Verthash, Autolykos, Octopus and FishHash have none on small prizes (rows 8, 21, 22, 26). The honest reading: the random program on a DRAM-latency bound is the strongest construction the history has, and nobody has tested it at Ethereum's prize.
5. **Periodic human forks fail as a defence.** Monero: four forks in 20 months; chips were back at 85% of the hashrate within four months of the v8 fork (row 16), and Vorick wrote that a chip able to survive forks at under a 5x hit had been designed. Vertcoin: three forks, two followed by rented-hash 51% attacks within weeks, because each fork reset the hashrate to a rentable size (row 7). Ravencoin: FPGA bitstreams for the new order within weeks (row 9). Sia: the fork bricked competitors' chips and left one vendor at 37% (row 14). Grin: the tweaks worked because the lane was scheduled to die (row 18). The chip's design cycle is 5 months for Bitmain (Vorick) and 13 for a startup; a fork every 6 months is a race the chip wins on the second lap, and each fork is a governance event. Igneum's draws are automatic and scheduled at genesis; that is the right side of this lesson, and section 4.3 asks whether the epoch can also be shorter than a bitstream (addition 5).
6. **RandomX got the target right and the derivation right, and it costs GPUs 25x.** Right: the work is "code and data" so a fixed circuit cannot serve it; chained programs defeat filtering (0.44x); the dataset is above any SRAM die; the item derivation is a random superscalar program tuned to DRAM latency so the light-mode chip pays as much energy per item as a DRAM read (section 2.4). The cost: a GPU runs the VM at 25x worse per joule than a CPU (row 17), which is the cost Igneum refuses, and the reason the program is per hour and compiled. What Igneum did not take: the random item derivation (addition 2) and four external audits before launch (addition 3).
7. **ProgPoW got the datapath right and lost on governance and one unpriced attack.** Right: target the commodity hardware's whole datapath (random math, register file, cache reads, DAG loads) so a chip has to be a GPU; the hardware audit agreed for compute-only chips (1.1x to 1.2x, Rao). Unpriced: the DAG on die (Least Authority suggestion 2, Rao's "<< 0.1x"), the same attack Igneum calls M16. Not adopted: two tentative approvals, a petition, bugs found late (Kik), authorship disputes and a PoS roadmap; the change needed a contentious fork on a live chain (row 20). Igneum's lesson is the one it already follows: every layer goes in before the public testnet as a genesis rule or a reserve, so no adoption vote is ever needed.
8. **What a "GPU-friendly" chain lost when its GPU miner fell behind: the miners, then the chain's shape.** Kaspa's hashrate rose 7x in months and its GPU share went to nothing; seven forks left to re-resist (row 23). Alephium, Nervos, Handshake, Kadena and Radiant went the same way without the forks (rows 25, 27, 28, 15, 29). Iron Fish forked away from its own Blake3 within a year of the first box (row 25). The chains kept their security budget and lost the fleet that had launched them; the fleet's hardware went to the next GPU chain. For Igneum the metric is the share of hashrate on consumer cards by model, which is what the January 2027 benchmark should report and what the observer can estimate earlier (addition 4).
9. **An unreviewed memory-hard construction has a shortcut until someone looks.** MTP fell from 2 GB to under 1 MB before launch (row 11); Catena's proofs were flawed; Argon2i's parameters were attackable at the IRTF's "paranoid" setting; Cuckoo's memory claim was off by 50x within two months (section 2.3). Igneum's M_r and chained cache have had no cryptanalysis (`MEMHARD.md` section 3, ledger M7); the acceptance rule is a statistical filter, not a proof. The x8 decision multiplies the mixer's weight in the chip model, which multiplies the cost of a structural weakness in it (addition 3).
10. **The secret chip is found by its share, and it is on the chain before the announcement.** Monero's chips held 85% before anyone saw them; a nonce-pattern analysis found them (row 16). Zcash's Z9 was "5x to 10x below" what Obelisk's own study said the hash allowed, which is Vorick's evidence that better secret chips existed (section 1.1 row 5). A detector costs an observer query; a bounty costs escrow (addition 4).
## 4. The audit of Igneum against the history
### 4.1 The first-generation layers (live in class v2 and carried into v3)
| Layer | Answers which failure | Does not answer | Evidence |
|---|---|---|---|
| Random program per epoch from a VDF seed, 12 integer families, nonce-dependent select | Fixed-function chips (lesson 1); program filtering and seed grinding (RandomX's 0.44x, spec 04's 130-to-1) | A "GPU without graphics": a programmable sequencer over 12 ops and 8 registers (ledger M1); an FPGA overlay or bitstream compiled within the hour (rows 7, 9: FPGAs were the first adversary of Lyra2REv2 and X16R); the 7.5x AMD gap is a one-vendor fleet | Rows 9, 10, 16, 17, 20 |
| Weak-program acceptance, exact 16 loads, fresh-source rule | Per-program hash-rate spread (1.10x residual) that a chip could pick | Nothing it claims to; a chip's advantage cannot come from the program (status 20:16) | Census [I1] |
| 1 GiB to 2 GiB dataset of 4-byte random reads, latency-bound, cache 256 MiB | SRAM-scale memory (lesson 2); the bandwidth lever at the honest card (lesson 3: 128 x 4 B keeps the 5090 at 9% of its stream bandwidth) | The partial-store chip with a custom memory system (lesson 3); the time-memory curve (O-1.6) | Rows 1, 3, 16; [P13] |
| 8 dependent cache reads per item, fixed-shape mixer with drawn constants | The on-die recompute chip (Least Authority's light-evaluation attack), priced at 2.45x bare under v2 | The fixed shape gives the chip its 3x factor (lesson 6); no cryptanalysis (lesson 9) | [S69] [S70]; M16 |
| Era draws from chain state (op weights, fold rotations), reserve families by height, dataset growth, no human release | Fork fatigue and fork-reset attacks (lesson 5); the chip that "survives forks at under 5x" (Vorick) is the chip that the draws are meant to outlast | The draws touch the program, not the item derivation, so they cost the recompute chip nothing (`chip-model-v3.md` section 2) | Rows 7, 16, 18 |
| Warp-unit CPU verification without the dataset (2.1 ms per warp under x8) | Keeps the verifier light, the Equihash and Cuckoo goal | Caps every lever: the mixer budget stops at the 10 ms gate | [P16] [P19] |
### 4.2 The Counter ASIC 2.0 layers as decided tonight
| Layer | Decision (status file) | Answers | Does not answer | History's verdict |
|---|---|---|---|---|
| 1 Load width 4, 16, 64 B | Keep 4 B (w16 closes nothing) | Keeps the 5090 latency-bound (9% of stream) | The AMD 7.5x gap (2.4 G reads/s at every width) | Right by lesson 3; the vendor gap is a 3.0 question and a soft form of lesson 8 |
| 2 Per-program width mix | Out (spread over 5% on every card) | n/a | n/a | Right: a per-program spread is what a chip picks (lesson 1's X16R: randomised order gave 1.3x, the shape still fixed) |
| 3 Per-warp scratch with RMW | Out (does not move the recompute chip; 2.4x at every share; costs GPUs 12% to 48%) | n/a | n/a | Right: SRAM-tier work favours the chip (lesson 2; Rao: SRAM is the chip's weapon) |
| 4 + 8 Era layout (stride, interleave) and per-site windows | In (era inside the class) | A hard-wired layout tuned to one era | A programmable address decoder (era-layout.md section 8 says so); costs the recompute chip nothing | Small by itself; its value is in lesson 5 (automatic change without a fork) |
| 5 Hot table sized to GPU cache | Measured, not adopted (honest card pays g = 0.84 to 0.93; chip pays SRAM) | n/a | n/a | Right by lesson 2 |
| 6 Cache growth | Option C: doubles with the dataset (256 MiB, 512 MiB year 4, 1 GiB year 12) | Keeps the mirror on a leading node (section 2.5) | n/a | Right; RandomX's cache-to-dataset ratio of 8 is reached at year 4 |
| 7 INT8 matrix family | Reserve R1 = mm8, W_new 4, unlock era 4 or 90% signal | A family that a 12-op chip lacks | Matrix hardware is the most abundant custom silicon on earth; Least Authority's suggestion 5 was "watch ML hardware"; Apple's emulation costs 1.6x to 4.7x per op | Keep in reserve, order it last (addition 6) |
| 9 Epoch length as an era parameter (10 min to 2 h) | Reserve only, design on `ca2-epoch` | The bitstream-per-epoch FPGA (rows 7, 9) | The FPGA overlay (a soft GPU) and the HBM FPGA | Rank it up (addition 5) |
| Mixer x8 (M16's lever) | In: 0.31x bare, 0.92x with the 3x factor, verifier 2.1 ms per warp, daily build 23 to 77 ms | The on-die recompute chip (lesson 6, Least Authority's attack) | Its own fixed shape (the 3x factor stays) and its lack of review (lesson 9) | The right lever; additions 2 and 3 are what the history says to do to it next |
### 4.3 Ranked additions and upgrades
Ranked by how much the history says each would change the outcome, with the cost to GPUs and the risk. "Genesis" means a rule fixed before the public testnet; "reserve" means a named family or parameter in the genesis reserve, unlockable by height or 90% signal; "nowhere" means do not add.
| Rank | Addition | What it does | Evidence | Cost to GPUs | Risk | Where |
|---|---|---|---|---|---|---|
| 1 | **Price the partial-store chip and draw the time-memory curve.** A chip that stores a fraction f of the dataset in HBM or on many narrow DRAM channels, recomputes the rest from a 256 MiB on-die cache under x8, and reads with 4-byte granularity. Rows for f = 0.25, 0.5, 1 at HBM3 and at GDDR7 random-read rates, priced in energy per hash (Rao's metric) and in reads in flight per watt | The only chip class that beat a memory-bound GPU hash: Ethash's 2.1x to 4.8x came from the memory system with no on-die dataset (rows 3, 4); Rao priced a 16-die DAG holder under a GPU board in 2019; Cuckoo's curve was wrong by 50x until drawn [P20]; O-1.6 is open and `MEMHARD.md` section 3 item 2 says the curve was never drawn | None (analysis) | The row may come out over 2x, which would qualify the public claim before anyone else does | Genesis (before the vectors freeze) |
| 2 | **A random item-derivation program per day** in place of the fixed-shape mixer: a SuperscalarHash-style generator, integer only, drawn from the day key, with its own acceptance test, compiled once a day by miners and verifiers | RandomX's reason for SuperscalarHash: a fixed derivation is hard-wired by a chip; a random one makes the light-mode chip a CPU (section 2.4). In Igneum's model the fixed shape is the 3x factor that turns 0.31x into 0.92x; removing the factor is worth more than x8 to x16 would be (x16: 0.46x with the factor by M16's table) | None per hash (the daily build is 23 to 77 ms at x8 and would roughly double); the verifier needs a per-day compiled derivation (a JIT, or a round schedule drawn from a fixed set of reviewed rounds), measured against the 10 ms gate | Cryptanalysis of random ARX programs; weak draws; a JIT in the verifier is new attack surface; the vendors must agree bit-exactly on a program they compile | Reserve (named family, unlock by height or signal) now; genesis if the verifier cost is measured under the gate before the freeze |
| 3 | **External cryptanalysis of M_r, the chained cache and the acceptance rule before genesis**, with the x8 shape as the target | Lesson 9 (MTP, Catena, Argon2i, Cuckoo); RandomX bought four audits for $141,000 before launch [S60]; the x8 decision multiplies the mixer's weight in the chip model, so a shortcut inside the mixer is now worth 8x more to a chip | None | Finding something late moves the vectors; not finding it in time moves nothing | Genesis gate (ledger M7, raised in priority) |
| 4 | **The clock and the detector.** (a) A share-pattern detector on the observer: per-program hash-rate spread, nonce-group patterns and per-card-model rate bands, with an alert when a population behaves like one fixed design (MoneroCrusher's method); (b) a stated trigger: the bounty escrowed and the benchmark live before daily issuance crosses about $50K (Vorick's rule), not on a calendar date | Lesson 10 (85% secret share); section 2.5's table (chips at $20K to $30K a day on compute-bound hashes); D11 (the bounty is unfunded) | None | A detector with false positives; a trigger Josh has to fund | Not a layer; genesis-independent; do it before the public testnet |
| 5 | **Rank layer 9 (the epoch length) up, and measure the FPGA lane**: the compile-ahead cost per card at a 10-minute epoch (the `ca2-epoch` work), plus an estimate of a soft-overlay FPGA miner with HBM (reads in flight per watt against the 5090's 17.5 G/s) | FPGAs were the first adversary of Lyra2REv2 and X16R and came back within weeks of X16Rv2 (rows 7, 9); Xelis forked for FPGA resistance (row 30); a per-hour program is a bitstream target in a way a per-hash program is not | At 10-minute epochs: 6x the compile work per card (measured on `ca2-epoch`); the VDF lead shrinks | A short epoch moves the difficulty window (spec 1.12) and the seed path | Reserve (as decided), with the measurement before the public testnet |
| 6 | **Order the reserve by chip-unfriendliness**: families that force a full 32-bit datapath per lane first (byte permute, bit-field extract, variable shifts, popcount, select, the second shuffle form), mm8 last | Least Authority's "watch ML hardware"; int8 matrix blocks are licensable IP at every node; Apple pays 1.6x to 4.7x per emulated dot4 (status 20:38) | None at launch | None | Reserve ordering, genesis |
| 7 | **A vendor-share metric and a 3.0 target for the AMD gap**: the share of hashrate by vendor published with the benchmark, and the line-width question kept open as the plan says | Lesson 8: a one-vendor fleet is a softer version of chip capture; Equihash's NVIDIA tilt and Ethash's balance were part of each chain's miner politics (rows 3, 5) | n/a | A width that closes the gap makes the 5090 bandwidth-bound (status 20:27) | Counter ASIC 3.0 |
Checks the history suggests that are not layers:
| Check | Why | Source |
|---|---|---|
| 1. Header grinding for cache locality: can a miner search the pre-PoW header hash H for 32-lane groups whose 128 loads cluster into fewer DRAM rows or cache lines, at a search cost below the gain? | Kik's ProgPoW exploit and Dinur-Nadler's MTP attack were both "the attacker steers the addresses" | [S71] [P18] |
| 2. The chip detector's baseline: the per-program spread per card model, from the first week of the public testnet | Needed before addition 4(a) can alert | [S54] |
| 3. The 28 nm SRAM density figure in section 2.5 (approximate, from memory) and the node-cost consequence, cited properly | It is the argument that the cache size sets the chip's project cost | [E10] |
Evaluated and placed nowhere, with the reason:
| Candidate | Verdict | Reason |
|---|---|---|
| Program entropy per hash (RandomX) instead of per hour | Nowhere | Per-hash programs need an interpreter or JIT on the GPU, which is the 25x GPU penalty RandomX pays (row 17) and the reason Igneum compiles per epoch. The filtering attack per-hash chaining prevents is already closed by the VDF seed and the acceptance rule. Per-hour's residual exposure is the FPGA lane, which addition 5 addresses with a shorter epoch, not with per-hash programs |
| Superscalar dependency-chain design for the program itself | Nowhere, beyond what exists | The program's ALU work is not the defence (lesson 1); the dependency chain that matters is the 8 dependent cache reads per item and the 128 dependent loads per hash, both in place. The superscalar idea belongs in the item derivation (addition 2) |
| Verthash's table from the blockchain; a dataset derived from chain history | Nowhere | Against the recompute chip and the partial-store chip it changes nothing: both build the table from the same public inputs the GPU does. The day key already comes from a VDF of chain state, which gives the unpredictability without a history dependency; a history dependency costs the verifier the history (Verthash needs the headers) and ties the hash to pruning (spec 10) |
| Grin's dual PoW with a shifting split | Nowhere | It is a scheduled surrender (row 19). Igneum's automatic schedules (dataset growth, reserve unlocks, cache doubling) are the shifting split applied to one hash; a second lane would hand a chip a lane |
| Autolykos v1's non-outsourceability | Nowhere | It stops pools, not chips, and Ergo removed it after 19 months because contract pools bypassed it (row 21); Igneum needs pools (spec 09) |
| A per-hash VRF against nonce grinding | Nowhere | A signature per attempt is what NexaPow did and it got a 3x to 4x chip (row 24): EC arithmetic is fixed-function work. The grinding Igneum must guard is the header-locality search (check 1), which a VRF does not touch |
| Ternary or variable-precision integer ops | Nowhere, beyond the reserve | Every family must be bit-exact on three vendors; dot4 is native on NVIDIA and AMD and emulated on Apple at 1.6x to 4.7x (status 20:38), so each precision added is paid by the weakest vendor. The reserve already holds the integer-exact candidates; adding more does not change lesson 1 |
| Cache-timing-bound reads (ProgPoW's 16 KB cache, RandomX's L1 tier) | Nowhere | Measured out tonight at the L2 tier (layer 5) and the per-warp tier (layer 3): the honest card pays and the chip buys SRAM at $0.23 per MB. Rao's audit says the same about ProgPoW's cache reads |
| Divergent data-dependent branches | Nowhere (already excluded) | Branches cost a GPU divergence and a chip nothing; RandomX's single predictable branch targets speculative CPUs, which Igneum does not have |
| Floating point | Nowhere (already excluded) | Vendor rounding splits the chain (spec 1.14); RandomX could afford it because its target is one ISA family with IEEE semantics |
## 5. Decisions this raises for Josh
| # | Decision | Recommendation |
|---|---|---|
| 1 | Add the partial-store chip rows to `chip-model-v3.md` and draw the time-memory curve before the public testnet | Yes, before the vectors freeze (addition 1) |
| 2 | Name a random item-derivation program as a reserve family, and fund the verifier measurement that would move it to genesis | Reserve now; genesis if the verifier lands under the gate (addition 2) |
| 3 | Commission the external cryptanalysis of M_r and the chained cache before genesis, with the x8 shape as the target | Yes (addition 3; ledger M7) |
| 4 | Escrow the bounty and set its trigger to daily issuance, not to a date; build the share-pattern detector on the observer | Yes to the detector now; the escrow is Josh's (D11) |
| 5 | Rank the epoch-length reserve above the mm8 reserve, and measure the FPGA lane | Yes (additions 5 and 6) |
## 6. Sources and limits of this research
Research was gathered by four sub-agents between 20:10 and 20:45 UTC on 5 October 2026 and checked against the citations below. Fetch failures they reported: eprint.iacr.org PDFs sit behind a challenge page (abstract pages worked), so the Ren-Devadas energy figures, the Alwen-Blocki EuroS&P tables and the Lyra2 exponent come from abstracts; medium.com and bitcointalk.org returned 403 (Vorick's post was read through archive.sia.tech and secondary coverage; the IfDefElse posts through the Veil interview); Bitmain's prospectus PDF was blocked; the Dash iBeLink date and Octopus's "matrix step" are unverified; the 28 nm SRAM bit cell is from memory. Every hash-per-joule gain is derived from the cited rate and watt figures and is approximate.
Igneum sources: [I1] `docs/analysis/weak-program-census-2026-10-03.md`; `docs/plans/counter-asic-2.md` (be4b295); `docs/plans/counter-asic-2-status.md` and `docs/plans/counter-asic-2-rollout.md` (`ca2-coord`); `docs/analysis/chip-model-v3.md` (`ca2-mixer` 1ab8b21); `docs/analysis/m16-recompute-attacker-2026-10-05.md`; `docs/analysis/sram-mirror.md` revision 2 (`ca2-analysis`); `docs/plans/era-layout.md` (`ca2-era`); `docs/plans/hot-table.md` (`ca2-cache`); `docs/plans/read-width.md` (`readwidth`); `docs/spec/01-lottery-hash.md`, `04-seeds-and-vdf.md`; `docs/bench-log.md` ("the 9070 XT on the eGPU", `opencl-rdna4`); `proto-metal/MEMHARD.md`; `docs/fud-ledger.md` M1, M3, M7, M16, C2, D11.
History rows:
- [S1] https://medium.com/@Linzhi/what-is-memory-hard-45a363b59dfe (Tenebrix's 2011 claim); https://en.wikipedia.org/wiki/Litecoin
- [S2] https://jamesachambers.com/early-bitcoin-asic-miner-pictures-history/ (Gridseed); https://www.mikewesson.com/2013/04/29/mining-litecoin-on-ati-radeon-7970s/ (7970 at 700 kH/s)
- [S3] https://www.design-reuse.com/news/34403/innosilicon-28nm-litecoin-asic-reference-miner.html (A2, Apr 2014); https://www.coindesk.com/markets/2014/05/14/kncminer-reveals-additional-titan-scrypt-asic-specs (Titan)
- [S4] https://www.asicminervalue.com/miners/bitmain/antminer-l3-504mh ; https://cryptoage.com/en/2550-bitmain-antminer-l7-is-a-new-asic-miner-for-litecoin-and-dogecoin.html ; https://www.coindesk.com/markets/2014/09/11/dogecoin-community-celebrates-as-merge-mining-with-litecoin-begins
- [S5] https://docs.dash.org/en/stable/docs/user/introduction/features.html ; https://www.dash.org/news/happy-birthday-darkcoin/
- [S6] https://cryptomining-blog.com/7117-the-first-x11-mining-asic-ibelink-dm384m-asic-dash-miner/ ; https://cryptomining-blog.com/7493-power-usage-and-noise-of-the-ibelink-dm384m-x11-asic-miner/ ; https://bitcointalk.org/index.php?topic=854257.320 (R9 280X at 4 MH/s)
- [S7] https://99bitcoins.com/guides-and-tutorials/dash-mining/antminer-d3-review/ ; https://www.cryptocompare.com/mining/asic-miner-market/baikal-giant-x10-x11-10ghs/
- [S8] https://ethereum.org/developers/docs/consensus-mechanisms/pow/mining/mining-algorithms/dagger-hashimoto/
- [S9] https://cryptoslate.com/bitmain-e3-asic-ethereum-miner/ (4 Apr 2018: E3 4.44 W/MH against a tuned 1080 Ti and RX 570); https://hothardware.com/news/bitmain-launches-ethereum-asic-miner-hashrate-comparable-8-gtx-1080-gpus
- [S10] https://innosilicon.global/product/innosilicon-a10-pro-6gb-ethereum-miner-500-mh-s/ ; https://www.notebookcheck.net/The-NVIDIA-GeForce-RTX-3080-is-an-Ethereum-mining-monster-overclocked-cards-deliver-nearly-100-MH-s-double-the-Radeon-RX-5700-XT.494246.0.html
- [S11] https://www.coindesk.com/tech/2020/12/21/linzhi-begins-rollout-of-long-awaited-ethereum-miner-phoenix ; https://www.theblock.co/post/88622/questions-new-ethash-asic-ethereum (F2Pool: 2,733 MH/s at about 3,000 W)
- [S12] https://miningnow.com/asic-miner/jasminer-x4-2500mh-s/ ; https://www.asicminervalue.com/miners/bitmain/antminer-e9-2-4gh ; https://2miners.com/blog/asic-miners-for-ethereum-antminer-e3-vs-innosilicon-a10-eth-master-comparison/ (the 3% estimate)
- [S13] https://eips.ethereum.org/EIPS/eip-1057 ; https://www.theblock.co/news/ecosystems/2020-02-26-ethereum-community-members-submit-dissenting-progpow-petition-57061 ; https://ethereum.org/roadmap/merge/ ; https://cointelegraph.com/news/bitmains-antminer-e3-to-continue-mining-ether-with-new-update (the E3's 4 GB limit)
- [S14] https://ethereumclassic.org/blog/2020-11-27-thanos-hard-fork-upgrade/
- [S15] https://www.asicminervalue.com/miners/bitmain/antminer-e9-pro-3-68gh ; https://pool.kryptex.com/device/asic/jasminer/x16-p ; https://whattomine.com/coins/151-eth-ethash/asics
- [S16] https://eprint.iacr.org/2015/946 (Equihash); https://en.wikipedia.org/wiki/Zcash
- [S17] https://variance.hu/2017/05/08/748-solsec-zcash-equihash-teljesitmeny-egy-gtx-1080-ti-kartyabol/ (1080 Ti at 748 Sol/s)
- [S18] https://www.coindesk.com/markets/2018/05/03/bitmains-latest-crypto-asic-can-mine-zcash ; https://coinguides.org/innosilicon-a9-zmaster-50k-sols-equihash-asic/ ; https://support.bitmain.com/hc/en-us/articles/360012223994-Z9-Specifications ; https://www.asicminervalue.com/miners/bitmain/antminer-z11 ; https://d-central.tech/miners/antminer-z15/
- [S19] https://github.com/ZcashFoundation/zfnd/blob/master/_posts/blog/2018-05-08-statement-on-asics.md ; https://www.coindesk.com/tech/2018/06/28/zcash-votes-against-asic-resistance-in-boon-for-big-miners ; https://electriccoin.co/blog/ecc-roadmap-calls-for-focus-on-wallet-proof-of-stake-and-interoperability/
- [S20] https://blog.horizen.io/zencash-statement-on-double-spend-attack/ ; https://blog.horizen.io/horizen-zen-statement-on-mining-algorithm/ ; https://forum.zcashcommunity.com/t/list-of-all-coins-projects-on-equihash-asic-resistant-not-resistant/29085
- [S21] https://en.wikipedia.org/wiki/Bitcoin_Gold ; https://gist.github.com/metalicjames/71321570a105940529e709651d0a9765
- [S22] https://fluxofficial.medium.com/zels-custom-pow-algorithm-zelhash-activation-in-mid-june-ad3d14d72135 ; https://uploads-ssl.webflow.com/60c73eaed3399e074029d643/60fd7d883200fcde5ceb7049_ZelHash_v1.0.pdf
- [S23] https://github.com/BeamMW/beam/wiki/BEAM-Mining ; https://docs.beam.mw/BeamHashII.pdf ; https://medium.com/minerstat/beamhashiii-beam-forks-to-a-new-algorithm-at-block-777777-dd2aeacc9e5
- [S24] https://aion.theoan.com/blog/aion-mainnet-launch-kilimanjaro/ ; https://miningpoolstats.stream/zero
- [S25] https://vertcoin.io/history/ ; https://www.newsbtc.com/2014/12/01/vertcoin-introduces-new-pow-algorithm-promises-asic-free-features/
- [S26] https://cryptoage.com/en/1231-first-asic-miner-lyra2rev2-dayun-zig-z1.html ; https://www.asicminervalue.com/miners/dayun/zig-z1 ; https://whattomine.com/gpus/36-nvidia-geforce-gtx-1080-ti ; https://arxiv.org/pdf/1905.08792 (FPGA bitstreams)
- [S27] https://cryptobriefing.com/vertcoin-vtc-51-percent-attack/ ; https://en.wikipedia.org/wiki/Vertcoin ; https://www.fxstreet.com/cryptocurrencies/news/vertcoin-cryptocurrency-network-fell-victim-to-attack-51-201912030704
- [S28] https://github.com/vertcoin-project/vertcoin-core/releases/tag/0.14.0 (Lyra2REv3)
- [S29] https://soundcloud.com/vertcoin-talk/vertcoin-talk-episode-24-verthash-fork-happens-january-30th-2021 ; https://crazy-mining.org/en/software/wallets/vertcoin-vtc-instructions-for-mining-on-verthash/
- [S30] https://coincub.com/mining/how-to-mine-vertcoin-vtc/
- [S31] https://ravencoin.org/assets/documents/X16R-Whitepaper.pdf ; https://tronblack.medium.com/ravencoin-asic-thoughts-e6c0079609e6
- [S32] https://cryptoage.com/en/1782-asics-ow-miner-ow1-and-skc-miner-turing-r1-for-the-x16r-algorithm-exist.html ; https://en.cryptonomist.ch/2019/09/17/mining-ravencoin-hashrate/
- [S33] https://en.cryptonomist.ch/2019/10/02/ravencoin-rvn-hard-fork/ ; https://cryptomining-blog.com/11320-ravencoin-rvn-getting-fpga-mining-support-for-the-x16rv2-algorithm/
- [S34] https://medium.com/minerstat/kawpow-ravencoin-forks-to-a-new-algorithm-2e730cd09fb3 ; https://tronblack.medium.com/ravencoin-kawpow-expectations-a6a063df58f2
- [S35] https://github.com/RavenProject/Ravencoin/blob/master/roadmap/README.md ; https://whattomine.com/coins/234-rvn-kawpow/gpus ; https://miningreturns.com/learn/ravencoin-mining-guide
- [S36] https://www.neoxa.net/whitepaper/ ; https://woolypooly.com/en/blog/ravencoin-algorithm
- [S37] https://arxiv.org/pdf/1606.03588 (Egalitarian computing, MTP)
- [S38] https://eprint.iacr.org/2017/497 (Dinur and Nadler); http://blog.zorinaq.com/attacks-on-mtp/ ; https://firo.org/2017/07/21/mtp-audit-and-implementation-bounty.html
- [S39] https://firo.org/2018/12/05/mtp-faq-all-you-need-to-know.html
- [S40] https://firo.org/2021/10/01/firopow-and-instantsend-release.html
- [S41] https://firo.org/2025/11/19/hardfork-successful-nov-2025.html
- [S42] https://docs.decred.org/research/blake-256-hash-function/
- [S43] https://www.asicminervalue.com/miners/innosilicon/d9-decredmaster ; https://www.asicminervalue.com/miners/obelisk/dcr1 ; https://crypto.news/hardware-companies-are-launching-dedicated-asic-miners-for-decred/ (DCR1 at 28 nm)
- [S44] https://cryptoage.com/en/1254-bitmain-antminer-dr3-7,8-th-s-on-the-algorithm-blake-14r-decred.html ; https://medium.com/luxor/bitmain-antminer-dr5-decred-setup-guide-1c417f5f61fc
- [S45] https://1stminingrig.com/antminer-a3-review-bitmain-surprises-everyone-with-this-new-siacoin-miner/ ; https://medium.com/obelisk-blog/obelisk-update-may-june-2018-260fce12a825 ; https://www.eastshoremining.com/tutorial-innosilicon-s11-siamaster-3-83th-siacoin-miner/
- [S46] https://www.coindesk.com/markets/2018/10/19/sia-network-releases-hard-fork-code-to-block-crypto-mining-giants ; https://siasetup.info/learn/forks
- [S47] Vorick, "The state of cryptocurrency mining", 13 May 2018: https://archive.sia.tech/the-state-of-cryptocurrency-mining-538004a37f9b (read through https://davidgerard.co.uk/blockchain/2018/05/14/from-sia-an-incendiary-post-on-the-state-of-cryptocurrency-mining-in-2018/ and https://zycrypto.com/asic-manufacturer-shares-important-information-for-token-creators-and-miners/); Bitmain's reply https://blog.bitmain.com/en/bitmain-sia-state-cryptocurrency-mining/
- [S48] https://medium.com/kadena-io/kadena-public-blockchain-releases-fully-public-testnet-v3-hashing-algorithm-and-mining-api-e230a51c7b26 ; https://www.coindesk.com/markets/2019/11/04/kadena-goes-live-announces-new-token-sale-aiming-for-20-million
- [S49] https://www.asicminervalue.com/miners/goldshell/kd5 ; https://asicmarketplace.com/product/goldshell-kd2-kadena-miner-6-4-th-s/ ; https://asicmarketplace.com/product/bitmain-antminer-ka3-kadena-miner-166th/
- [S50] https://minerstat.com/hardware/nvidia-rtx-3080-lhr
- [S51] https://bytecoin.org/old/whitepaper.pdf ; https://docs.getmonero.org/proof-of-work/cryptonight/ ; https://en.wikipedia.org/wiki/CryptoNote
- [S52] https://news.8btc.com/bitmain-to-release-antminer-x3-cryptonight-asic-miner-with-220-khs-hashrate ; https://bitcointalk.org/index.php?topic=3127974.0 ; https://www.asicminervalue.com/miners/baikal/bk-n ; https://cointelegraph.com/news/bitmain-announces-new-monero-mining-antminer-x3-cryptos-devs-say-will-not-work
- [S53] https://github.com/monero-project/monero/pull/3253 (v7); https://coinguides.org/monero-network-upgrade-v8-cnv2-beryllium-bullet/ ; https://github.com/SChernykh/CryptonightR (CN-R: chip latency up 2.5x)
- [S54] https://medium.com/@MoneroCrusher/analysis-more-than-85-of-the-current-monero-hashrate-is-asics-and-each-machine-is-doing-128-kh-s-f39e3dca7d78 ; https://beincrypto.com/hashrate-analysis-reveals-asics-account-for-85-of-monero-mining/
- [S55] https://github.com/tevador/randomx (30 Nov 2019)
- [S56] https://github.com/tevador/RandomX/blob/master/doc/design.md
- [S57] https://github.com/tevador/RandomX/blob/master/doc/specs.md
- [S58] https://xmrig.com/benchmark/5kFcJv (3950X); https://whattomine.com/coins/101-xmr-randomx/gpus (RTX 3090 at 2.0 kH/s, 290 W)
- [S59] https://bt-miners.com/products/bitmain-antminer-x5-monero-miner-212k-bt-miners/ ; https://bitmain.com.vc/news/bitmain-launches-antminer-x9 ; https://pineconeinibox.shop/product/pinecone-matches-inibox-r1x-xmr-edition/ ; https://github.com/xmrig/xmrig/blob/master/doc/ALGORITHMS.md ; https://rfc.tari.com/RFC-0131_Mining ; https://www.theblock.co/post/353240/tari-privacy-network-merged-monero-mining-launch-mainnet
- [S60] https://github.com/tevador/RandomX/blob/master/README.md (the four audits and their cost); https://blog.trailofbits.com/2019/07/02/state/
- [S61] https://github.com/mimblewimble/docs/blob/master/docs/about-grin/proof-of-work.md ; https://github.com/tromp/cuckoo/blob/master/README.md ; https://github.com/tromp/cuckoo/blob/master/doc/cuckoo.pdf
- [S62] https://forum.grin.mw/t/mid-july-pow-hardfork-cuckaroo29-cuckarood29/5082 ; https://www.cudominer.com/grin-network-update-hard-fork-16th-january-2020/ ; https://forum.grin.mw/t/grin-v5-0-0-network-upgrade-hard-fork-4-january-2021/7895
- [S63] https://docs.aeternity.com/aeternity-core-concepts/protocol/consensus-mechanisms/cuckoo-cycle-proof-of-work ; https://medium.com/cortexlabs/miners-can-now-test-mine-on-testnet-dolores-in-preparation-for-the-mainnet-launch-cf851d7b0146
- [S64] https://forum.grin.mw/t/introducing-the-grn1-a-cuckatoo31-asic-from-obelisk/2519 ; https://medium.com/obelisk-blog/grn1-cancellation-announcement-54782c6e3e83
- [S65] https://bitcointalk.org/index.php?topic=5219851.0 ; https://forum.grin.mw/t/innosilicons-grin-asics-canceled/6932 ; https://ipollo-miners.com/product/ipollo-g1/
- [S66] https://forum.grin.mw/t/another-cuckatoo-bounty-succesfully-claimed/11739 (Apr 2025)
- [S67] https://eips.ethereum.org/EIPS/eip-1057 ; https://github.com/ifdefelse/ProgPOW
- [S68] https://github.com/ethereum/pm/blob/master/AllCoreDevs-EL-Meetings/Meeting%2052.md ; https://www.coindesk.com/markets/2019/01/04/ethereum-developers-give-tentative-greenlight-to-asic-blocking-code ; https://souptacular.github.io/2020-03-02-progpow-the-ethereum-community-speaks/ ; https://www.coindesk.com/tech/2020/03/06/ethereums-progpow-call-features-frustration-but-little-progress
- [S69] https://leastauthority.com/static/publications/LeastAuthority-ProgPow-Algorithm-Final-Audit-Report.pdf (9 Sep 2019)
- [S70] https://github.com/ethcatherders/progpow-audit ("Bob Rao - ProgPOW Hardware Audit Report Final.pdf", Sep 2019)
- [S71] https://github.com/kik/progpow-exploit (4 Mar 2020); https://github.com/Souptacular/linzhi (Linzhi's 3x to 8x claim)
- [S72] https://cryptoage.com/en/1238-bitcoin-interest-bci-and-new-mining-algorithm-progpow.html ; https://en.wikipedia.org/wiki/Zano_(blockchain_platform) ; https://github.com/sero-cash/serominer ; https://x.com/QuaiNetwork/status/1880037240149057759 ; https://veil-project.com/blog/2020-OhGodAGirl/
- [S73] https://docs.ergoplatform.com/mining/autolykos/ ; https://ergoplatform.org/en/blog/2019_07_09_after_launch/ ; https://bytwork.com/en/news/khardfork-ergo-07
- [S74] https://www.hashrate.no/gpus/3090/ERG
- [S75] https://mining.confluxnetwork.org/ ; https://github.com/Conflux-Chain/conflux-rust
- [S76] https://github.com/Conflux-Chain/CIPs/blob/master/CIPs/cip-102.md
- [S77] https://www.kaspafaq.com/sp_accordion_faqs/what-is-kheavyhash/ ; https://github.com/Dagmbisrat/Kaspa-FPGA-Miner
- [S78] https://whattomine.com/coins/352-kas-kheavyhash/gpus/49-nvidia-geforce-rtx-3090
- [S79] https://www.cryptominerbros.com/product/iceriver-ks0-100gh-s-kas-miner/ ; https://www.asicminervalue.com/miners/iceriver/ks1 ; https://apextomining.com/product/new-bitmain-antminer-ks3-8-3t-3188w-kas-miner-asic-mining-machine-profitable-comining-soon/ ; https://www.asicminervalue.com/miners/bitmain/antminer-ks5-pro-21th
- [S80] https://hashdag.medium.com/kaspa-where-to-part-iv-last-c68717a8d309 (May 2023); https://miningreturns.com/news/kaspa-asic-mining-era-what-you-need-to-know
- [S81] https://x.com/karlsennetwork/status/1829148683104870534 ; https://www.hashrate.no/c/Algorithm_change_for_Karlsen_and_Pyrin ; https://github.com/spectre-project/rusty-spectre ; https://cryptix-network.org/whitepaper ; https://network.hoosat.fi/public/htn-whitepaper-2.pdf ; https://bugna.org/
- [S82] https://spec.nexa.org/mining/NexaPOW/
- [S83] https://www.cryptominerbros.com/product/dragonball-miner-a21-nexa-miner/ ; https://whattomine.com/coins/357-nexa-nexapow
- [S84] https://docs.alephium.org/frequently-asked-questions/ ; https://medium.com/@alephium/one-year-of-mainnet-b7ed5d3024ee
- [S85] https://hashrate.no/gpus/3090/ALPH
- [S86] https://www.asicminervalue.com/miners/goldshell/al-box ; https://mineshop.eu/bitmain-antminer-al1 ; https://www.zeusbtc.com/Asic-Miner/Asic-Miner-Details.asp?ID=3719
- [S87] https://fips.ironfish.network/fips/fip-3-memory-hard-mining-algorithm ; https://fips.ironfish.network/fips/fip-10-hardfork-1 ; https://github.com/iron-fish/fish-hash ; https://github.com/karlsen-network/fish-hash-plus
- [S88] https://medium.com/nervosnetwork/a-decentralized-mainnet-launch-for-nervos-ckb-9cb119d15540
- [S89] https://www.asicminervalue.com/miners/bitmain/antminer-k5-1130gh ; https://www.asicminervalue.com/miners/goldshell/ck5 ; https://2miners.com/blog/nervos-ckb-network-hashrate-increased-asics-are-the-cause/
- [S90] https://www.coindesk.com/markets/2020/02/04/handshakes-uncensorable-web-domains-go-live-on-mainnet
- [S91] https://www.goldshell.com/news/goldshell-announces-best-handshakehns-miner-hs1-coming-soon/ ; https://www.asicminervalue.com/miners/goldshell/hs3
- [S92] https://radiantblockchain.org/ ; https://d-central.tech/miners/rxd-rx0/
- [S93] https://cryptoage.com/en/2929-video-card-hashrate-based-on-the-sha512-256d-algorithm-cryptocurrency-mining-radiant-rxd.html
- [S94] https://cryptomining-blog.com/11865-mining-zano-using-the-progpowz-proof-of-work-algorithm/
- [S95] https://github.com/dynexcoin/DynexSolve ; https://minerstat.com/coin/DNX/faq
- [S96] https://docs.warthog.network/janushash/ ; https://github.com/CoinFuMasterShifu/Janushash
- [S97] https://docs.xelis.io/network-upgrades
- [S98] https://docs.verus.io/overview/verus-proof-of-power.html
Papers:
- [P1] https://www.microsoft.com/en-us/research/publication/moderately-hard-memory-bound-functions/
- [P2] https://www.wisdom.weizmann.ac.il/~naor/PAPERS/mem.pdf
- [P3] https://www.tarsnap.com/scrypt/scrypt.pdf
- [P4] https://eprint.iacr.org/2014/238
- [P5] https://eprint.iacr.org/2016/115
- [P6] https://eprint.iacr.org/2016/759
- [P7] https://eprint.iacr.org/2016/989
- [P8] https://www.cryptolux.org/images/d/d0/Argon2ESP.pdf
- [P9] https://eprint.iacr.org/2016/027
- [P10] https://eprint.iacr.org/2015/227
- [P11] https://eprint.iacr.org/2013/525
- [P12] https://eprint.iacr.org/2015/136
- [P13] https://eprint.iacr.org/2017/225
- [P14] https://eprint.iacr.org/2018/221
- [P15] https://eprint.iacr.org/2017/443
- [P16] https://eprint.iacr.org/2015/946
- [P17] https://arxiv.org/abs/1606.03588
- [P18] https://eprint.iacr.org/2017/497
- [P19] https://eprint.iacr.org/2014/059
- [P20] https://da-data.blogspot.com/2014/03/a-public-review-of-cuckoo-cycle.html ; http://www.cs.cmu.edu/~dga/crypto/cuckoo/analysis.pdf
- [P21] https://arxiv.org/abs/1902.00112
- [P22] https://ieeexplore.ieee.org/document/8516911/
- [P23] http://www.vldb.org/pvldb/vol13/p898-feng.pdf
- [P24] https://arxiv.org/abs/2002.11064
Economics and silicon:
- [E1] https://europractice-ic.com/schedules-prices-2025/
- [E2] https://newsletter.semianalysis.com/p/the-dark-side-of-the-semiconductor (24 Jul 2022)
- [E3] https://semiengineering.com/big-trouble-at-3nm/ (21 Jun 2018); https://semiengineering.com/the-increasingly-uneven-race-to-3nm-2nm/ (24 May 2021); https://semiengineering.com/what-will-that-chip-cost/ (30 Oct 2023)
- [E4] https://siliconanalysts.com/analysis/fabless-startup-tapeout-cost-guide (1 Mar 2026, secondary)
- [E5] https://patentpc.com/blog/chip-manufacturing-costs-in-2025-2030-how-much-does-it-cost-to-make-a-3nm-chip (secondary, approximate)
- [E6] https://techcrunch.com/2018/09/26/bitmain-hong-kong-ipo/ ; https://bitcoinmagazine.com/markets/bitmain-ipo-prospectus-reveals-offering-may-be-gamble-investors ; https://www.chaincatcher.com/en/article/2057998
- [E7] https://www.sec.gov/Archives/edgar/data/1780652/000119312519297270/d773846d424b4.htm
- [E8] https://michaeltaylor.org/papers/bitcoin_taylor_cases_2013.pdf ; https://michaeltaylor.org/papers/Taylor_Bitcoin_IEEE_Computer_2017.pdf
- [E9] https://www.tomshardware.com/news/amd-unveils-more-ryzen-3d-packaging-and-v-cache-details-at-hot-chips (Aug 2021); https://www.graphcore.ai/posts/introducing-second-generation-ipu-systems-for-ai-at-scale ; https://www.theregister.com/software/2020/09/29/groq-is-hard-to-grok-but-reckons-its-ai-chips-roq-ex-googlers-unorthodox-design-now-shipping-to-customers/1170931
- [E10] https://newsletter.semianalysis.com/p/tsmcs-3nm-conundrum-does-it-even (21 Dec 2022); https://fuse.wikichip.org/news/7343/iedm-2022-did-we-just-witness-the-death-of-sram/ ; https://www.tomshardware.com/news/no-sram-scaling-implies-on-more-expensive-cpus-and-gpus ; https://images.nvidia.com/aem-dam/Solutions/geforce/blackwell/nvidia-rtx-blackwell-gpu-architecture.pdf (GB202: 128 MB L2 on the full die, 96 MB on the RTX 5090, 750 mm^2)
- [E11] CoinMarketCap historical snapshots, https://coinmarketcap.com/historical/YYYYMMDD/ for the dates in the section 2.5 table (pulled 5 Oct 2026); HBM pricing https://www.trendforce.com/presscenter/news/20240506-12125.html and https://www.nextplatform.com/2024/02/27/he-who-can-pay-top-dollar-for-hbm-memory-controls-ai-training/ ; the E3's DDR3 and the A10's presumed GDDR6 from the ProgPoW FAQ https://medium.com/@ifdefelse/progpow-faq-6d2dce8b5c8b (Jan 2019, read through secondary coverage) and https://coingeek.com/memory-limitations-prompt-bitmain-antminer-e3-to-halt-etc-support/
Latency:
- [L1] https://terpconnect.umd.edu/~blj/papers/memsys2018-dramsim.pdf
- [L2] https://arxiv.org/abs/1712.08304
- [L3] https://docs.nvidia.com/nsight-compute/ProfilingGuide/index.html ; https://developer.download.nvidia.com/video/gputechconf/gtc/2020/presentations/s21819-optimizing-applications-for-nvidia-ampere-gpu-architecture.pdf
- [L4] https://arxiv.org/abs/2604.15751

View file

@ -0,0 +1,85 @@
# Card lifetime per tier: how many years a card keeps mining
5 October 2026. Consequences review, sub-agent of the consequences reviewer. Desk arithmetic only; nothing was run.
## 1. Inputs
| Input | Source | Value used |
|---|---|---|
| Dataset schedule | `docs/spec/01-lottery-hash.md` 432 to 437 | 2 GiB at genesis plus 0.5 GiB a year (2,048 + 512 x years MiB) |
| Index mapping (a) | same file, line 442 | multiply-shift: the dataset grows every day, continuous |
| Index mapping (b) | same file, line 442 | power-of-two steps 2, 4, 8 GiB on the schedule's average: 4 GiB at year 4, 8 GiB at year 12; my extrapolation: 16 GiB at year 28, 32 GiB at year 60 |
| Scratch per resident warp | `igneum-wt-ca2-cache/docs/plans/hot-table.md` 66 to 73 | 32 or 128 KiB per warp; 5090 = 170 SMs x 48 warps = 8,160 (approximate, from memory) |
| Hot table, buffers | same file, 70 | hot table 32, 64 or 96 MiB (96 used here); buffers 128 MiB |
| Cache | hot-table.md 70 (resident, 256 MiB in every total) against `igneum-wt-ca2-era/docs/plans/era-layout.md` 93 ("resident only while the day's dataset is built, then free") | both readings carried: resident = worst case, freed = best case. The two plans disagree and gate 1 should say which |
| Cache growth | `igneum-wt-ca2-coord/docs/plans/counter-asic-2-status.md` 79 (layer 6 option C) | 256 MiB at genesis, 512 MiB at year 4, 1 GiB at year 12; by the same rule 2 GiB at year 28, 4 GiB at year 60 |
| Budget rule | same file, 17: the whole working set stays under 6 GB on an 8 GB card | my reading: 75% of card memory at every tier. Apple: 50% of unified memory, because macOS, the display and the node share it; that share is my assumption |
| Public claims | `site/index.html` 443, 461; `site/litepaper.html` 560; `docs/evidence.md` | quoted in Table 3. evidence.md has no row on card lifetime |
Card memory is binary (8 GB = 8,192 MiB). The hot-table row "An 8 GB card at 5090 occupancy" (line 73) counts 8,160 warps; a real 8 GB card has 20 to 24 SMs, so its scratch is about a tenth of that row. Resident warps below are SMs x 48 (NVIDIA Ampere and later), SM counts from memory, approximate; Apple uses the 2,048 warps the Metal harness launches (hot-table.md 66).
## 2. Table 1: non-dataset working set per tier (MiB)
Worst = scratch 128 KiB, cache resident. Columns g / y4 / y12 = genesis, year 4, year 12 (the cache doublings). Freed = era-layout's reading, constant over the years.
| Tier | Card assumed (SMs, approximate) | Warps | Scratch 128 KiB | Scratch 32 KiB | Cache resident, 128 KiB: g / y4 / y12 | Cache resident, 32 KiB: g / y4 / y12 | Cache freed: 128 / 32 KiB |
|---|---|---|---|---|---|---|---|
| 4 GB | GTX 1650 (14 SMs x 32 warps, Turing) | 448 | 56 | 14 | 536 / 792 / 1,304 | 494 / 750 / 1,262 | 280 / 238 |
| 8 GB | RTX 3050 (20) | 960 | 120 | 30 | 600 / 856 / 1,368 | 510 / 766 / 1,278 | 344 / 254 |
| 12 GB | RTX 3060 (28) | 1,344 | 168 | 42 | 648 / 904 / 1,416 | 522 / 778 / 1,290 | 392 / 266 |
| 16 GB | RTX 5060 Ti (36) | 1,728 | 216 | 54 | 696 / 952 / 1,464 | 534 / 790 / 1,302 | 440 / 278 |
| 24 GB | RTX 4090 (128) | 6,144 | 768 | 192 | 1,248 / 1,504 / 2,016 | 672 / 928 / 1,440 | 992 / 416 |
| 32 GB | RTX 5090 (170) | 8,160 | 1,020 | 255 | 1,500 / 1,756 / 2,268 | 735 / 991 / 1,503 | 1,244 / 479 |
| Apple 8 to 64 GB | M-series, harness launch count | 2,048 | 256 | 64 | 736 / 992 / 1,504 | 544 / 800 / 1,312 | 480 / 288 |
Every row = scratch + 96 (hot table) + 128 (buffers) + cache (256 / 512 / 1,024 when resident). The freed reading still peaks at dataset + cache during the daily build, but that peak is smaller than the resident total whenever hashing pauses for the build, so the freed column is the steady-state set.
## 3. Table 2: dataset room and the year the dataset outgrows it
Room = usable memory (75%, Apple 50%) minus Table 1. Worst = 128 KiB scratch, cache resident (room shrinks at years 4, 12, 28, 60). Best = 32 KiB scratch, cache freed. Option (a): the year 2,048 + 512 x y exceeds the room. Option (b): the first step the room cannot hold; the card mines up to that day.
| Tier | Usable MiB (share) | Room at genesis, worst / best | (a) ends, years, worst / best | (b) ends, year, worst / best |
|---|---|---|---|---|
| 4 GB | 3,072 (75%) | 2,536 / 2,834 | 1.0 / 1.5 | 4 / 4 |
| 8 GB | 6,144 (75%) | 5,544 / 5,890 | 6.3 / 7.5 | 12 / 12 |
| 12 GB | 9,216 (75%) | 8,568 / 8,950 | 12.0 / 13.5 | 12 / 28 |
| 16 GB | 12,288 (75%) | 11,592 / 12,010 | 17.1 / 19.5 | 28 / 28 |
| 24 GB | 18,432 (75%) | 17,184 / 18,016 | 28.0 / 31.2 | 28 / 60 |
| 32 GB | 24,576 (75%) | 23,076 / 24,097 | 37.6 / 43.1 | 60 / 60 |
| Apple 8 GB | 4,096 (50%) | 3,360 / 3,808 | 2.6 / 3.4 | 4 / 4 |
| Apple 16 GB | 8,192 (50%) | 7,456 / 7,904 | 10.1 / 11.4 | 12 / 12 |
| Apple 32 GB | 16,384 (50%) | 15,648 / 16,096 | 25.1 / 27.4 | 28 / 28 |
| Apple 64 GB | 32,768 (50%) | 32,032 / 32,480 | 55.1 / 59.4 | 60 / 60 |
What the table says per tier:
| Tier | Reading |
|---|---|
| 4 GB | Mines at genesis with 488 to 786 MiB spare. Under (a) it is out within 1 to 1.5 years. Under (b) it lasts to the year-4 step, as the spec's own remark says (line 442) |
| 8 GB | 6 to 7.5 years under (a). 12 years under (b): "more than a decade" is true only under (b), and only just |
| 12 GB | The year-12 cache doubling (1 GiB resident) is what ends it, under both options, if the cache stays resident. With the cache freed it reaches year 28 under (b). This tier's lifetime is decided by the cache residency question, not by the dataset |
| 16 GB | 17 to 19.5 years under (a), year 28 under (b) |
| 24 GB | Under the resident reading the year-28 cache doubling (2 GiB) ends it the same day under both options. Freed: 31 years or year 60 |
| 32 GB | 38 to 43 years under (a), year 60 under (b). Not a constraint for any plan |
| Apple 8 GB | 2.6 to 3.4 years under (a), year 4 under (b). The base 8 GB Apple laptop is a short-lived miner |
| Apple 16 GB | 10 to 11.4 years under (a), year 12 under (b): the same shape as an 8 GB card |
| Apple 32 / 64 GB | 25 years and 55 years or more. No constraint |
Proving is a separate budget (the 15.6 GB peak the 12 GB mine-and-prove question came from); this file covers mining only.
## 4. Table 3: the public sentences against the numbers
| Where | Sentence now | What the tables give | Proposed sentence (Josh decides the wording) |
|---|---|---|---|
| `site/index.html` 443 | Memory: "2 GB, fixed" (RandomX) / "2 GB, growing" (Igneum) | 2 GiB at genesis, plus 0.5 GiB a year on average under either option | "2 GB, growing 0.5 GB a year". The row is right; the rate is the useful addition |
| `site/index.html` 461 | "Any 4 GB card, approximate." | True at genesis (2,584 to 2,834 MiB of a 3,072 MiB budget). Ends at 1 to 1.5 years under (a), year 4 under (b) | "Any 4 GB card at launch, 8 GB for the long run, approximate." |
| `site/litepaper.html` 560 | "a 4 GB card mines for about four years and an 8 GB card for more than a decade, approximate." | 4 GB: 1 to 1.5 years (a) or 4 years (b). 8 GB: 6.3 to 7.5 years (a) or 12 years (b). Both numbers hold only under option (b) | If gate 1 picks (b): "a 4 GB card mines until the first dataset step at year 4, an 8 GB card until the second at year 12 and a 16 GB card until year 28, approximate." If (a): "a 4 GB card mines for about a year, an 8 GB card for about seven and a 16 GB card for about seventeen, approximate." |
| `site/litepaper.html` 560 | "12 GB or more proves full shards." | Not a lifetime claim; left as is. For mining, 12 GB lasts 12 years with the cache resident, year 28 with it freed under (b) | No change from this file |
| `docs/evidence.md` | No row on card lifetime | The litepaper sentence is a public claim with no row | Add a row, label "designed", sources: spec 1.13.3 and this file; status moves to "tested" once a 4 GB and an 8 GB card run the genesis working set under the cap |
## 5. Reading
- The two index-mapping options end on the same day where a cache doubling takes the last of the room. With the cache resident that is the 12 GB tier at year 12 and the 24 GB tier at year 28 (Table 2, worst column). Under option (b) every tier ends on a step day by construction, so a tier ends on the same day under both options exactly when option (a) also ends it on a doubling day.
- Everywhere else option (b) is kinder: 4 GB gains about 2.5 years, 8 GB about 5, 16 GB about 10. The site and litepaper numbers are option (b) numbers. If gate 1 picks (a), both public sentences are wrong today by 2.5 to 5 years.
- The cache residency disagreement (hot-table.md 70 against era-layout.md 93) decides the 12 GB tier's lifetime (12 against 28 years) and nothing else. It should be settled at gate 1 beside the mapping choice.
- The 75% rule is my generalisation of "under 6 GB on an 8 GB card"; at 4 GB it leaves 1 GB for the driver and the display, which a headless rig would not need. A 4 GB card on a bare Linux rig might hold out to year 2 under (a). Not measured.

View file

@ -0,0 +1,100 @@
# The on-die-cache recompute chip against the RTX 5090, class v2 and class v3, everything combined
5 October 2026 (night), Counter ASIC 2.0, worker ca2-mixer. The model is M16's
(`docs/analysis/m16-recompute-attacker-2026-10-05.md`): the strongest chip the plan has priced holds the whole
cache in SRAM and derives every dataset item instead of reading it, so its cost per hash is item derivations,
and its rate at a 50 T op/s integer budget (an RTX 5090's, approximate) is `50 T / (ops per hash)`. Nothing here
is a measurement of a chip; every GPU figure says where it was measured. "Approximate" marks a figure from memory.
## 1. Inputs
| Input | Value | Source |
|---|---|---|
| Items per hash | 128 (one item per load, 128 loads per hash, median 128.00 distinct) | spec 01 sections 1.4.2 and 1.8.5; the 20,000-program census |
| Integer operations per mixer application | about 130 | spec 01 section 1.8.4 |
| Mixer applications per item | 9 under v2; 36 under v3 (`m = 4`, `docs/plans/mixer-x4.md`) | `memhard::Shape::mixers_per_item` |
| Integer operations per item | 1,170 (v2); 4,680 (v3) | 9 x 130; 36 x 130 |
| Integer operations per hash | 149,760 (v2, "150,000"); 599,040 (v3, "600,000") | 128 x the above |
| Chip integer budget | 50 T op/s (approximate: 21,760 ALUs at about 2.4 GHz, one 32-bit operation each per clock) | M16 section 3 |
| Fixed-function factor | 3x (approximate, from memory: 2x to 5x is the usual credit for a pipeline with no scheduling or divergence) | M16 section 3 |
| RTX 5090, version 2 programs, measured | 136.1 MH/s (readwidth, tonight, `docs/plans/read-width.md`, pack w4 on PC 2); 139.7 MH/s (M11, 4 October, `docs/bench-log.md`) | this analysis uses tonight's 136.1 as the denominator and quotes both |
| RTX 5090 at w16 (16-byte loads), measured | 139.8 MH/s | readwidth table, tonight (the width stays 4 B: w16 closes nothing) |
| Cache mirror, 256 MiB, N5 headline density | 128 mm^2, $46 per good die (64 mm^2, $21 at the bit-cell lower bound) | `docs/analysis/sram-mirror.md` revision 2, sections 4 and 5 (`ca2-analysis` e6085c6) |
| Cache mirror plus a 96 MB hot table, N5 headline | 175 mm^2, $68 | same, so a hot table costs 0.49 mm^2 and $0.23 per MB (linear, approximate) |
| 512 MiB and 1 GiB mirrors, N5 headline | 255 mm^2 and 510 mm^2; $111 to $306 | same, section 4 (the growth rule's cache at years 4 and 12, priced at today's node) |
| GPU-class die | 750 mm^2 (the equal-silicon comparison) | M16 section 3 |
| CPU verifier, one M5 Max core (loaded, load average 5.6; ratios are the measurement) | v2 1.31 to 1.36 ms per unit, x4 1.92 to 1.96 (1.45x), x8 2.79 (2.1x); worst cold 1.58 / 2.04 / 2.94 ms | `docs/plans/mixer-x4.md` section 6.4, 5 October 2026 21:40 UTC |
## 2. The rows
Chip rate = 50 T op/s / ops per hash. "Bare" = chip rate / 136.1 MH/s. "With the factor" = bare x 3. "Equal
silicon" = bare x (750 - SRAM) / 750 x 3: the SRAM takes die area the logic does not get, the M16 convention
("minus the area the SRAM takes"). SRAM in mm^2 and dollars at the N5 headline density.
| Row | Mixer | Ops per hash | Chip rate at 50 T op/s | SRAM the chip holds | mm^2 / $ (N5 headline) | Bare gain against 136.1 MH/s | With the 3x factor | Equal silicon, SRAM deducted, with the factor |
|---|---|---|---|---|---|---|---|---|
| v2 as shipped (the M16 and scratch-soundness row) | x1 | 149,760 | 334 MH/s | 256 MiB | 128 / $46 | 2.45x (2.39x against 139.7) | 7.4x | 6.1x |
| v2 at w16 (not adopted; the chip's cost is items, not bytes: unchanged) | x1 | 149,760 | 334 | 256 MiB | 128 / $46 | 2.39x against 139.8 | 7.2x | 5.9x |
| x4 (the candidate measured beside v3; not v3) | x4 | 599,040 | 83.5 MH/s | 256 MiB | 128 / $46 | 0.61x | 1.84x | 1.53x |
| MEASURED, NOT ADOPTED (layer 5 decided out of v3 on the PC rows, coordinator 21:40 UTC): v3 plus a 32 MiB hot table, added form (16 dataset loads and k hot loads): the honest card pays the hot loads, this chip pays SRAM only | x4 | 599,040 (a hot load is one SRAM read, no item) | 83.5 | 288 MiB | 144 / $53 | 0.66x at the Mac's g = 0.93 (126.6 MH/s); 0.70x at the 5090's g = 0.87 (118.4); the 9070 XT's g 0.84 | 1.98x (Mac g), 2.11x (5090 g) | 1.60x, 1.71x |
| MEASURED, NOT ADOPTED: v3 plus a 64 MiB hot table, added form | x4 | 599,040 | 83.5 | 320 MiB | 160 / $61 | 0.71x at the Mac's g = 0.87 (118.4 MH/s); 0.73x at the 5090's g = 0.84 (114.3); the 9070 XT's g 0.80 | 2.12x (Mac g), 2.19x (5090 g) | 1.67x, 1.73x |
| v3 at year 4 (cache 512 MiB under option C, dataset 4 GiB), no hot table | x4 | 599,040 | 83.5 | 512 MiB | 255 / $111 | 0.61x | 1.84x | 1.21x |
| v3 at year 12 (cache 1 GiB, dataset 8 GiB) | x4 | 599,040 | 83.5 | 1 GiB | 510 / $306 | 0.61x | 1.84x | 0.59x |
| **v3: mixer x8** (decided 22:05 UTC under the delegated rule: verify 2.1 ms per unit on one Mac core against the 10 ms gate, the daily 1 GiB build 23 to 77 ms on the 5090 and the 9070 XT) | x8 | 1,198,080 | 41.7 | 256 MiB | 128 / $46 | 0.31x | 0.92x | 0.76x |
| x8 at year 4 | x8 | 1,198,080 | 41.7 | 512 MiB | 255 / $111 | 0.31x | 0.92x | 0.61x |
The era draws of spec 1.13.1 cost the chip nothing in this model: the mixer round count is not drawn, the op
weights and fold rotations change the program, not the item derivation, so the chip's ops per hash stand. The
width rule (4-byte loads kept) changes nothing either: w16 would have moved the honest denominator by 2.7% and the
chip's cost not at all.
Arithmetic, row v3: 36 x 130 = 4,680 ops per item; x 128 = 599,040 per hash; 50 x 10^12 / 599,040 = 83.5 x 10^6
hashes per second; 83.5 / 136.1 = 0.613; x 3 = 1.84; equal silicon (750 - 128) / 750 = 0.829, x 1.84 = 1.53.
Hot table rows: 32 MiB x 0.49 mm^2 per MB = 16 mm^2, 64 MiB = 32 mm^2 (the 96 MB column of `sram-mirror.md`
scaled linearly); (750 - 144) / 750 = 0.808 and (750 - 160) / 750 = 0.787. The honest denominator in the added
form is the v2 rate times `g`, the card's measured ratio with the hot loads added: on the M5 Max tonight
`g = 0.93 / 0.87 / 0.83` at 32 / 64 / 96 MiB (the cache agent, relayed by the coordinator at 21:23 UTC;
`docs/plans/hot-table.md` carries the runs); the 5090's and the 9070 XT's `g` are the PC rows, owed, and until they
land the row carries the Mac's `g` against the 5090's rate, which is a mixed figure and is marked so. Year 4 and 12 rows: the mirror of
`sram-mirror.md` section 4 at N5 for 512 MiB and 1 GiB plus the 64 MiB table, at today's density (the node of
those years is denser by about 1.8x at year 10 on the trend the same file cites; the row is a floor on the area,
not a forecast).
## 3. The margin, plainly
The combined headline row is the mixer row alone (layer 5 is out: the added form costs the 5090 13 to 16 percent
and the 9070 XT 16 to 20 percent against the 0.97 bar, coordinator 21:40 UTC; the width stays 4 bytes; the era
draws and the cache growth cost this chip nothing at year 0), and class v3 is x8 (decided 22:05 UTC). The headline:
**the on-die-cache recompute chip at 50 T op/s reaches 41.7 MH/s against the 5090's 136.1, 0.31x bare, 0.92x with
the 3x fixed-function factor, 0.76x with the mirror's area deducted: under 1x with the factor, 0.92x, a margin of 8
percent on the factor (a 3.3x factor reads 1.0x) and of 9 percent on the budget (55 T op/s reads 1.0x).** The x4
candidate, measured beside it, read 1.84x and 1.53x. The hot-table rows above are kept as measured, not adopted:
against THIS chip an added hot table is a cost to the honest card and none to the chip, so it would have moved the
row the wrong way by the card's own `g`. The margin, plainly:
- the 3x fixed-function factor is approximate and from memory; at 3.3x the equal-budget row reads 2.0x;
- the denominator is one card's measured rate on one night (136.1 against 139.7 the night before: 2.6% apart);
- the 50 T op/s budget is approximate; a chip at 55 T op/s reads 2.0x;
- the hot table in the added form lowers the honest denominator by whatever the hot loads cost the GPU (owed from
the PC rows), which raises the chip's gain by the same share, 1.84x or more if the hot loads are free, higher if
not; the hot table's only cost to this chip is 16 to 32 mm^2 of die.
What keeps it under 1x is the mixer, and nothing else in Counter ASIC 2.0 moves this chip (the scratch at any share
gave 2.4x, `docs/analysis/scratch-soundness.md` section 3.4; the hot table taxes the DRAM-only chip, not this one;
the cache growth taxes it only in die area, which is cheap at year 0 and real at year 12). The next levers, in
order:
1. Mixer x16 (the next step of the same lever): 0.16x bare and 0.46x with the factor against 136.1; the verifier
by the measured increments (+0.63 ms at x4, +1.46 at x8 on the M5 Max core: about +3.1 ms at x16, 3.7 ms per
unit, 9 ms on a 2.5x slower laptop core, approximate) is at the edge of the 10 ms gate, so a 2019-class laptop
core measurement (O-1.14) decides it, not this model.
2. The hot table: adopted or not on the PC rows (`docs/plans/hot-table.md`); in the added form it costs the GPU
7 to 17 percent on the Mac and the chip die area only, so against this chip it is a lever in the wrong
direction and against a DRAM-only chip the first lever; if it is adopted, the mixer must carry the extra `1/g`
(x8 at g = 0.87 reads 1.06x at the equal budget, 0.84x with the SRAM deducted).
## 4. What this does not settle
The items of M16 section 5 stand: the inline kernel on NVIDIA with a 64 MiB cache inside L2 (a measured point
under the "50 T op/s" row) is a PC job not yet run; the time-memory curve (O-1.6) is not drawn; the mixer has had
no cryptanalysis, and a shortcut inside it cuts the 4,680 directly; no chip has been priced beyond its SRAM.

View file

@ -0,0 +1,175 @@
# Layer 7: the integer matrix family (INT8 x INT8 into INT32) as a reserved instruction family, design
5 October 2026 (night), Counter ASIC 2.0 (`docs/plans/counter-asic-2.md`, layer 7), branch `ca2-analysis`. Design
only: nothing here touches the generator, a vector or a node. Every figure is cited (vendor document, URL, section) or
measured (machine, date, command) or labelled approximate.
## 1. The primitive per vendor, from the vendor documents
| Vendor, hardware | Per-lane dot4 (4 bytes x 4 bytes into a 32-bit integer) | Warp or wave matrix (int8 tiles, int32 accumulate) | Source |
|---|---|---|---|
| NVIDIA, sm_61 and later (Pascal on) | PTX `dp4a.atype.btype d, a, b, c` with `.atype = .btype = {.u32, .s32}`: "Four-way byte dot product which is accumulated in 32-bit result"; semantics `d = c; for i in 0..3: d += Va[i] * Vb[i]` with the bytes sign- or zero-extended by type; introduced in PTX ISA 5.0, "Requires sm_61 or higher". CUDA: `__device__ int __dp4a(int srcA, int srcB, int c)` ("Four-way signed int8 dot product with int32 accumulate") and the unsigned form, plus `char4`/`uchar4` overloads | `mma.sync` with `.u8`/`.s8` A and B and `.s32` C and D: shape `.m8n8k16` "requires sm_75 or higher" (Turing on, PTX 6.5); shapes `.m16n8k16` and `.m16n8k32` require sm_80 (Ampere on, PTX 7.0); sparse `.m16n8k32` and `.m16n8k64` with `.u8`/`.s8` also exist | PTX ISA 9.4, section 9.7.1.24 (dp4a) and 9.7.16.5 (mma), https://docs.nvidia.com/cuda/parallel-thread-execution/index.html ; CUDA Math API, integer intrinsics, https://docs.nvidia.com/cuda/cuda-math-api/cuda_math_api/group__CUDA__MATH__INTRINSIC__INT.html ; read 5 October 2026 |
| AMD RDNA 3 (gfx11) | `v_dot4_i32_iu8` (VOP3P; each operand signed or unsigned by a per-operand bit, optional clamp) reached from clang/HIP/OpenCL C as `__builtin_amdgcn_sudot4(bool a_signed, int a, bool b_signed, int b, int acc, bool clamp)` (LLVM feature `dot8-insts`: "Has v_dot4_i32_iu8, v_dot8_i32_iu4 instructions"); `v_dot4_u32_u8` as `__builtin_amdgcn_udot4` (`dot7-insts`: "Has v_dot4_u32_u8, v_dot8_u32_u4"); `v_dot4_i32_i8` as `__builtin_amdgcn_sdot4` (`dot1-insts`: "Has v_dot4_i32_i8 and v_dot8_i32_i4"). gfx11's common feature set carries dot7, dot8, dot9, dot10 and dot12 | `V_WMMA_I32_16X16X16_IU8`: `__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32` and `_w64` (feature `wmma-256b-insts`), a 16x16x16 tile per wave | LLVM `clang/include/clang/Basic/BuiltinsAMDGPU.td` (main, read 5 October 2026), lines defining `__builtin_amdgcn_sdot4`, `udot4`, `sudot4`, `wmma_i32_16x16x16_iu8_w32`; AMD GPUOpen, "How to accelerate AI applications on RDNA 3 using WMMA", https://gpuopen.com/learn/wmma_on_rdna3/ ; the RDNA 3 ISA PDF itself did not download tonight (AMD's CDN refused curl and the fetcher timed out), so the instruction names are from the compiler and GPUOpen, not quoted from the ISA guide |
| AMD RDNA 4 (gfx12, the 9070 XT) | the same `sudot4` and `udot4` builtins: LLVM's `FeatureISAVersion12_Generic` carries `FeatureDot7Insts` and `FeatureDot8Insts` and not `FeatureDot1Insts`, so `__builtin_amdgcn_sdot4` is NOT exposed on gfx12 and `sudot4` with both operands signed is the signed form to use | `__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32_gfx12` and `_w64_gfx12` (feature `wmma-128b-insts`): the int8 WMMA exists on RDNA 4 with a narrower per-lane operand (2 ints per lane for A and B against 4 on RDNA 3); AMD's RDNA 4 WMMA guide names the same builtin | LLVM `llvm/lib/Target/AMDGPU/AMDGPU.td` (`FeatureISAVersion12_Generic`) and `BuiltinsAMDGPU.td` (main, 5 October 2026); AMD GPUOpen, "WMMA guide for AMD RDNA 4 architecture GPUs, part 2", https://gpuopen.com/learn/wmma-guide-amd-rdna-4-gpus-part-2/ ; the RDNA 4 ISA guide (AMD document 70651, April 2025) was not readable tonight (docs.amd.com returned 401 to a direct fetch) |
| AMD CDNA 3 (MI300) | the same VOP3P dot instructions (approximate: not checked in the CDNA 3 guide tonight) | `V_MFMA_I32_16X16X32_I8` and `V_MFMA_I32_32X32X16_I8` (opcodes 87 and 86 in the VOP3P-MFMA table) | AMD Instinct MI300 CDNA 3 ISA Reference Guide, 5 August 2025, https://www.amd.com/content/dam/amd/en/documents/instinct-tech-docs/instruction-set-architectures/amd-instinct-mi300-cdna3-instruction-set-architecture.pdf (downloaded and grepped 5 October 2026) |
| AMD, OpenCL on Adrenalin (Windows) | `cl_khr_integer_dot_product` is NOT in the 24 extensions Adrenalin lists for gfx1201 (the list: fp64, the int32 and int64 atomics, 3d image writes, byte addressable store, fp16, gl sharing, amd device attribute query, amd media ops and media ops2, d3d10, d3d11 and dx9 sharing, image2d from buffer, subgroups, gl event, depth images, mipmap image and writes, amd copy buffer p2p); the platform is OpenCL 2.1 so the OpenCL C 3.0 feature macro `__opencl_c_integer_dot_product_input_4x8bit` is not expected. What IS reachable: the Adrenalin OpenCL compiler is clang (driver string `PAL,LC`), and `__builtin_amdgcn_sudot4` from OpenCL C has been shown to emit `V_DOT4_I32_IU8` on a Radeon 780M (gfx1103, RDNA 3, driver 32.0.31041, Windows 11) at 2.7x the scalar fallback (1.27 to 3.42 TMAC/s) | not from OpenCL C | Adrenalin 26.9.2 extension list, https://geeks3d.com/20260904/amd-radeon-adrenalin-26-9-x-graphics-driver/ ; the OpenCL C route: https://github.com/1640675651/CPPminer/pull/1 (third party, one machine; the 9070 XT run of this document's probe is the check) ; `cl_khr_integer_dot_product` itself: OpenCL C 3.0 specification section 6.2.2.16, `int dot(char4, char4)` and `int dot_acc_sat(char4, char4, int)`, https://registry.khronos.org/OpenCL/specs/3.0-unified/html/OpenCL_Ext.html |
| Apple, Metal (MSL 4.1, 4 June 2026) | none. MSL has no dp4a or packed byte dot product: the built-in `dot(T x, T y)` is a geometric function on floating-point vectors (section 6.9); the integer functions of section 6.4 have no dot form. A per-lane dot4 is scalar emulation (section 4 below measures it) | `simdgroup_matrix<T, 8, 8>` exists for T = half, bfloat (Metal 3.1 and later) and float only (section 2.4: "T is half, bfloat ... or float"); no integer SIMD-group matrix. BUT Metal 4's tensor operation `mpp::tensor_ops::matmul2d` (section 7.2.1, table 7.3, "MatMul2D data type supported") lists A `char` x B `char` into C `int` (Metal 4) and `uchar` x `uchar` into `int` (Metal 4 and OS 26.4), plus `char` x `int4b_format` into `int`. So Apple has an exact int8 x int8 into int32 matrix path, on tensors (device or threadgroup memory, or a `cooperative_tensor` per SIMD-group or threadgroup), not on registers, and only through Metal 4's tensor API. Which GPU families run it in hardware (the M5's neural accelerators) against emulation is in the Metal Feature Set Tables, which the spec defers to and which were not read tonight | Metal Shading Language Specification version 4.1, https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf , sections 2.4, 6.9, 7.2.1 table 7.3 (PDF downloaded and text-extracted 5 October 2026) |
The Apple finding, stated plainly: the brief's expectation ("Apple has no int8 matrix or dot path") is half right. There
is no per-lane dot4 and no integer `simdgroup_matrix`. There is an exact `char x char -> int` matmul2d in Metal 4
(table 7.3). Two things about it are unverified tonight and matter for conformance: whether the int accumulate wraps or
saturates (the spec text I extracted says nothing either way; a vector at the int32 edge on the M5 settles it), and the
feature-set table (which Apple GPUs run it natively). What is settled: a generator op that is a per-lane dot4 has no
Apple intrinsic and costs scalar emulation; a generator op that is a whole-unit 8x8x16 or 16x16x16 int8 tile has a
native path on all three vendors (mma.sync on sm_75+, WMMA on RDNA 3 and 4, matmul2d on Metal 4), with Apple's path
living in a different API shape (tensors, not register fragments).
## 2. The family's semantics as a generator op (integer only, bit-exact)
Two forms are proposed; the reserve can hold both as separate families or one.
### 2.1 `dot4`: per-lane
```
dot4 dst = dst + dot4_u8(src, src2)
where dot4_u8(a, b) = sum over i in 0..3 of byte_i(a) * byte_i(b), bytes zero-extended, sum modulo 2^32
```
- Bytes are UNSIGNED. Reason, measured below: on Apple the unsigned emulation costs 1.6 ALU-chain steps per dot4
and the signed one 4.7 (section 4), while NVIDIA (`dp4a.u32.u32`) and AMD (`V_DOT4_U32_U8`, `udot4`, `dot7-insts`)
carry the unsigned form natively as they carry the signed one. Signed bytes buy nothing for the hash (the input is a
pseudo-random register) and cost the vendor without the intrinsic 3x more.
- Accumulation wraps modulo 2^32 like every other op in section 1 (spec 1.14 item 5). The maximum dot of four unsigned
bytes is 4 x 255 x 255 = 260,100, so no single dot4 overflows; the wrap is in the running sum, which is why the
AMD `clamp` bit and the OpenCL `dot_acc_sat` form are NOT the primitive (saturation would change results).
- Operands: `dst`, `src`, `src2` with `src != dst` as for `mad`; `src2` may equal either.
- Verifier: one closed-form integer expression per lane; the register-major interpreter of 1.11 adds four byte
multiplies and adds per lane. The CPU reference in the probes (`dot4_ref` in `proto-opencl/dot4-probe.c`) is this
expression.
### 2.2 `mm8`: the 32-lane unit as one int8 tile
The 32 lanes of a unit (spec 1.9) hold, in `src`, a 4-byte row fragment of an 8 x 16 int8 matrix A and, in `src2`,
a 4-byte column fragment of a 16 x 8 int8 matrix B, in exactly the layout of PTX `mma.m8n8k16` with `.u8` operands
(PTX ISA 9.4 section 9.7.16.5, "Matrix Fragments for mma.m8n8k16", the integer-type layout):
```
lane l (0..31): A[row = l >> 2][k = 4 * (l & 3) .. 4 * (l & 3) + 3] = the 4 bytes of src (byte 0 = lowest k)
B[k = 4 * (l & 3) .. +3][col = l >> 2] = the 4 bytes of src2
result C[r][c] = sum over k in 0..15 of A[r][k] * B[k][c] (uint8 x uint8, 16 products, exact, at most 1,040,400)
mm8 dst = dst + C[l >> 2][2 * (l & 3) + bit] bit = an immediate 0 or 1 drawn by the generator
```
Every lane receives one of the two C elements its lane position owns in the PTX fragment (`c0` for bit 0, `c1` for
bit 1), added into `dst` modulo 2^32. The whole op is a function of the unit's `src` and `src2` across all 32 lanes,
like `shfl`, so it needs the unit to be exactly 32 logical lanes (the wave64 rule of 1.9 applies: a wave64 device
holds two units and the local-memory path is used).
How each vendor runs it:
| Vendor | Native form | Cost per `mm8` (approximate until measured) |
|---|---|---|
| NVIDIA sm_75+ | one `mma.sync.aligned.m8n8k16.row.col.s32.u8.u8.s32` per warp, A and B fragments straight from `src` and `src2`, C = 0 in, `c0`/`c1` out, one add | one tensor instruction plus one add |
| AMD RDNA 3 and 4 | one `V_WMMA_I32_16X16X16_IU8` per wave32 with the 8x16 and 16x8 tiles zero-padded into 16x16 (the WMMA fragment layout differs from PTX's: a fixed permutation of bytes between lanes, which is a few `ds_bpermute` or `v_perm` operations, bit-exact) | one WMMA plus the permutation and the pad |
| AMD CDNA | `V_MFMA_I32_16X16X32_I8` with padding | as above |
| Apple, Metal 4 | `matmul2d<descriptor(8, 8, 16)>` on `uchar` A and B into an `int` cooperative tensor (table 7.3 row "uchar, uchar, int", OS 26.4), the fragments written from registers into a threadgroup tensor first (32 lanes x 8 bytes = 256 bytes), the C element read back per lane | one tensor op plus two threadgroup round trips; on Apple GPUs without the neural accelerators the runtime's emulation, unmeasured |
| Any vendor, fallback | 16 scalar byte products per lane after gathering the 16 bytes of B's column from the 4 lanes that hold them (4 shuffles or one 64-byte threadgroup exchange) | 4 shuffles plus 4 `dot4` emulations: on Apple about 4 x 1.6 = 6.4 ALU steps plus the shuffles (approximate, from the probe) |
Verifier: the unit evaluates C as 8 x 8 x 16 = 1,024 unsigned byte products once per `mm8` instruction and hands
each lane its element. That is 1,024 multiply-adds per instruction per unit, against 64 x 8 = 512 instructions per
hash: at W_new = 4 (section 3) a program carries about 2.6 `mm8` per iteration, 21 per hash, 21,500 multiply-adds per
unit per hash, under 10 microseconds on one core (approximate), far inside the 0.63 ms the verifier already spends per
unit (spec 1.11). The simulation stays exact because every product and sum is an integer with a defined wrap.
### 2.3 Which form to reserve
`mm8` is the one that takes matrix hardware at GPU scale from a chip (the plan's layer 7 row): a chip without tensor
units pays 1,024 products per unit per instruction where a GPU pays one tensor instruction. `dot4` is a per-lane ALU op
that a chip matches with four 8-bit multipliers, which is cheap silicon; it adds little chip resistance and costs Apple
emulation. Recommendation: reserve `mm8`; keep `dot4` out, or in only as `dot4_u8` behind `mm8`.
## 3. The genesis reserve entry (spec text for 1.13.2)
Proposed wording, to go under 1.13.2 as the first named reserve family once the conformance runs of section 5 pass:
> Reserve family R1, `mm8` (integer matrix). Semantics: section 2.2 of `docs/analysis/int8-matrix-family.md`,
> uint8 operands from `src` and `src2` in the m8n8k16 fragment layout, one int32 element of C per lane selected by
> the immediate `bit`, added into `dst` modulo 2^32. Weight at unlock `W_new = 4` points, taken proportionally from the
> ten live non-load families (the load weight and count are untouched, 1.13.1). Edge vectors, each a hand-built unit
> run on every vendor: all bytes 0xFF in A and B (C = 16 x 65,025 = 1,040,400 everywhere); all bytes 0x80 (C = 16 x
> 16,384 = 262,144); A all zero (C = 0); `dst` = 0xFFFFFFFF with a nonzero C (the wrap); alternating 0x00 and 0xFF by
> lane (the fragment mapping: C[r][c] nonzero only where the row and column bytes meet); `bit` = 0 and 1 on the same
> fragments. Unlock: at the start of era n = 4 (two years after genesis, DAA 62,208,000), or earlier by the 90%
> signalling path of section 5.7; never by a release.
The era-4 choice is deliberate: two years is long enough for the three vendors' tensor paths (and Apple's Metal 4
feature-set coverage) to be in every miner's driver, and short enough to land before any chip built against the
launch instruction set has paid back (approximate; a chip programme is 12 to 24 months, approximate, from memory).
Reserve rule for a vendor that can only emulate. Spec 1.13.2 as written requires conformance on every vendor; it says
nothing about cost. Proposed addition:
> A family enters the reserve when it is bit-exact on every vendor of 1.15. A vendor that reaches the result only by
> emulation (no instruction or library path) does not block entry if the measured penalty of the emulation on that
> vendor, on the family's own probe (a dependent chain of the op, G ops/s against the same vendor's integer ALU chain),
> is at most 8x per op, AND the family's weight at unlock keeps the emulating vendor's hash-rate loss under 5% on the
> memory-hard hash (the hash is latency-bound, so a per-op penalty on 4% of the instructions is a small fraction of a
> hash whose time is 128 dependent DRAM reads; the 5% is checked on the vendor's card with the family live, not
> computed). A family whose emulation exceeds either bound stays out of the reserve until the vendor ships a path.
With tonight's numbers: on Apple the unsigned `dot4` emulation is 1.6x per op (inside the bound); the signed one 4.7x
(inside, but why pay it); `mm8` through Metal 4's matmul2d is a path, not an emulation, and its cost is owed.
## 4. dp4a-class throughput, measured so far
Probe: a dependent chain of one dot4 per step per lane (`acc = dot4(x, y, acc); x = x * K + acc; y = rotl(y, 7) ^
(acc + s)`), 1,048,576 lanes x 4,096 steps, best of 3, device time, bit-exact against a CPU reference on two lanes
per run, beside the ALU chain of the 9070 XT bench-log entry (`x = x * K + rotl(y, 7); y = (y ^ x) + s`, 5 ops per
step counted). Sources: `proto-metal/dot4-probe.swift` (Metal), `proto-opencl/dot4-probe.c` (OpenCL: scalar, the
`cl_khr_integer_dot_product` `dot`, AMD `__builtin_amdgcn_sudot4`, NVIDIA inline PTX `dp4a.s32.s32`),
`proto-cuda/dot4-probe.cu` (CUDA `__dp4a` and the scalar emulation, for a PC with nvcc). Each OpenCL variant is built on
its own and a variant the platform cannot compile prints a "build failed" row.
| Card, API | Date, command | ALU chain, G steps/s | dot4 signed emulation, G dot4/s | dot4 unsigned emulation, G dot4/s | dot4 intrinsic, G dot4/s | Penalty of the emulation per op (ALU steps per dot4) | ok (bit-exact) |
|---|---|---|---|---|---|---|---|
| Apple M5 Max, Metal | 5 October 2026 20:0x UTC, `with-lock.sh measure ./dot4-probe` (swiftc -O), GPU start-to-end time | 879.8 (4.882 ms) | 188.2 (22.82 ms) | 548.2 (7.834 ms) | none exists | signed 4.7x, unsigned 1.6x | yes, all three kernels |
| Apple M5 Max, Apple OpenCL 1.2 | same, `with-lock.sh measure ./dot4-probe-cl --device 0`, event time | 871.5 (4.928 ms) | 188.4 (22.80 ms) | not in this probe | `cl_khr_integer_dot_product` not listed; the kernel using `dot(char4, char4)` compiled anyway and ran at 846 G/s but MISMATCHED the CPU reference on every lane checked (Apple's `dot` on char4 is not an integer dot; the extension macro must gate it) | signed 4.6x | alu and dot4e yes; dot4_khr NO |
| RTX 5090 (PC 1, ae432dc7), NVIDIA OpenCL 3.0 CUDA, driver 617.14 | 5 October 2026 20:29 UTC, job `run-dot4-20261005` (`relay/playbooks/dot4-probe.ps1`, both mining cards switched off in the app first, restored after; `node tools/jobs.mjs run-dot4-20261005`), event time | 8,753.5 (0.491 ms) | 1,239.1 (3.466 ms) | not in the OpenCL probe | 7,453.6 (0.576 ms) via inline PTX `dp4a.s32.s32` | emulation 7.1x; the intrinsic 1.17x (the chain is one dp4a plus 3 ops against 5 ops), so the emulation costs 6.0x the instruction | yes, all three |
| RX 9070 XT (PC 1, gfx1201, eGPU), AMD OpenCL 2.0 AMD-APP 3683.0 (PAL,LC) | same job, same time | 701.4 (6.124 ms) | 480.8 (8.932 ms) | not in the OpenCL probe | 664.3 (6.465 ms) via `__builtin_amdgcn_sudot4(true, a, true, b, acc, false)`: the Adrenalin OpenCL C compiler accepts the clang builtin and emits `v_dot4_i32_iu8` | emulation 1.46x; the intrinsic 1.06x, so the emulation costs 1.38x the instruction | yes, all three; the older 3652.0 platform entry for the same card gave 696.2 / 501.7 / 683.6 |
| Ryzen 9800X3D gfx1036 (PC 1, integrated RDNA 2, 2 CUs) | same job | 40.6 (105.9 ms) | 15.8 (272.3 ms) | | `sudot4` does not build: "needs target feature dot8-insts" (RDNA 2 has `dot1-insts`' `v_dot4_i32_i8`, the `sdot4` builtin, which the probe did not try) | emulation 2.6x | alu and dot4e yes |
| Every PC device | `cl_khr_integer_dot_product` not listed on NVIDIA (OpenCL 3.0) or AMD (2.0); the pragma draws "unknown OpenCL extension" on both and the `dot(char4, char4)` kernel does not build | | | | | | |
Reading across the three cards. Per dot4 at the hardware rate: the 5090 does 7.45 T dot4/s (one `dp4a` per step, 0.85
of its ALU-chain step rate), the 9070 XT 0.66 T (0.95 of its ALU-chain rate), the M5 Max 0.55 T at best (the unsigned
emulation; no instruction). On the ALU chain the 5090 is 12.5x the 9070 XT and 10x the M5 Max; on hardware dot4 it is
11.2x the 9070 XT, so the family does not widen the AMD gap, and 13.6x the M5 Max, so Apple's emulation widens its gap
by 1.4x on this op (approximate: one probe shape, the ratios of best-of-3 numbers). The signed emulation is where the
vendors differ most: 7.1x the ALU step on NVIDIA, 4.7x on Apple, 1.46x on AMD (AMD's compiler and byte-permute
hardware make the four sign-extended products nearly free; the NVIDIA OpenCL compiler does not pattern-match the
emulation into `dp4a`, which the 6x gap between `dot4e` and `dot4_nv` shows). None of this is a hash-rate number: the
hash is bound by 128 dependent DRAM reads, and a family at W_new = 4 adds about 21 of these ops per hash per lane
against about 1.2 microseconds of memory latency per hash per lane (approximate), so the per-op penalties above turn
into hash-rate losses well under 5% on every card, to be measured with the family live.
Reading of the Mac numbers. The ALU chain's 880 G steps/s on the M5 Max is the integer baseline (5 ops per step
counted, so about 4.4 T int ops/s, approximate; the 5090's 8,754 G steps/s is about 43.8 T, against the whitepaper's
104.8 peak INT32 TOPS which counts a multiply-add as two). A signed dot4 emulated as `int4(as_type<char4>(a))` products costs
4.7 of those steps; the unsigned form 1.6 steps. The 3x gap between the two is the sign extension (Metal lowers the
unsigned byte extraction to masks that fold into the multiplies, approximate reading of the result, not of the
compiled code). Both are far under the 8x bound of section 3, and the hash spends its time on DRAM reads, so a per-lane
`dot4` family would cost Apple a few percent at W_new = 4 (to be measured with the family live, not computed). The
Apple OpenCL `dot(char4, char4)` mismatch is the kind of thing the edge vectors of section 3 exist to catch.
## 5. What is owed or unverified
| Item | State |
|---|---|
| dp4a throughput on the RTX 5090 through NVIDIA OpenCL inline PTX | measured (section 4); the CUDA `__dp4a` form (`proto-cuda/dot4-probe.cu`) is unrun (no nvcc job tonight) and is a cross-check, not a gap |
| `sudot4` on the 9070 XT through Adrenalin's OpenCL C; `cl_khr_integer_dot_product` on the 3683.0 platform | measured: the builtin works and emits the instruction; the extension is not listed and the `dot(char4, char4)` kernel does not build |
| `sdot4` (`dot1-insts`) on RDNA 2 (gfx1036) | not tried; the probe only carries `sudot4` |
| Metal 4 `matmul2d` uchar x uchar into int on the M5 Max: wrap or saturate at the int32 edge, native or emulated, throughput | owed (a second Metal probe; the API needs a tensor set-up the dot4 probe does not have) |
| Metal Feature Set Tables: which Apple GPU families run int8 matmul2d natively | not read tonight |
| RDNA 3 and RDNA 4 ISA guides: the instruction text itself (names taken from LLVM and GPUOpen) | AMD's CDN refused the downloads tonight |
| `mm8` on AMD: the exact byte permutation between the PTX m8n8k16 fragment layout and the RDNA WMMA 16x16x16 layout | design, to be written with the kernel |
| The hash-rate cost of the family live at W_new = 4 on each vendor (the 5% rule of section 3) | owed, needs the generator change (not tonight) |
| Edge vectors of section 3 as files | owed, with the generator change |

View file

@ -0,0 +1,152 @@
# The prover floor: why a 12 GB card cannot prove on SP1 6.8.1's GPU server, and the patch
5 October 2026, 22:00 UTC on (Josh: "execute if it will solve the issue"). Branch `prover-floor`
(worktree `igneum-wt-prover-floor`). The measured facts this starts from: `docs/plans/proving-v1.md` and the
bench-log entry "proving v1" (branch proving-v1): the GPU server holds 13.9 GB for an empty shard, 20.4 GB at the
adopted v1 shard, 28.3 GB flat from 20 M to 60 M cycles, and no environment knob moved the floor. Every figure
below is from the source at tag v6.8.1 (cloned to `vendor/sp1-6.8.1`, gitignored; the fork is the patch
`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`) or from a PC 2 run named in `docs/bench-log.md`
("prover floor"). Sizes in GiB are computed from the source constants (4-byte field elements); sizes in MiB are
measured by `nvidia-smi` at 1 s.
## Where the server is built and what it reads
The SDK downloads `sp1_gpu_server_v6.8.1_x86_64.tar.gz` (133,750,780 bytes) from the SP1 release and runs it from
`$HOME/.sp1/bin/sp1-gpu-server` (`crates/cuda/src/server.rs` 19 to 30, 80 to 99). The source is in the same
repository: `sp1-gpu/crates/server` (the binary), built by `.github/workflows/release.yml` 234 to 314 on CUDA
12.8.1 with Go and protoc (`cargo build --release --bin sp1-gpu-server`). The binary takes no options
(`sp1-gpu/crates/server/src/main.rs` 15 to 18: `--version` only) and reads `CUDA_VISIBLE_DEVICES` (32 to 35);
everything else comes from the environment the host process passes it, through `SP1CoreOpts::default()`
(`crates/core/executor/src/opts.rs` 99 to 140: `SHARD_SIZE`, `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`,
`MINIMAL_TRACE_CHUNK_THRESHOLD`, `TRACE_CHUNK_SLOTS`, `FULL_SIZE_SHARDS`) and the worker counts
(`crates/prover/src/worker/config.rs`).
## The memory model, term by term
Every device buffer is sized at construction from constants, not from the shard. The server builds the prover at
the first `Setup` request (`sp1-gpu/crates/server/src/server.rs` 126 to 137) through
`cuda_worker_builder_with_machine` (`sp1-gpu/crates/prover_components/src/builder.rs` 102 to 148):
| Term | Where | Size | On the device | Moves with the shard |
|---|---|---|---|---|
| The gate | `builder.rs` 35 to 39: `gpu_memory_gb = ceil(total / GiB) + 4`; `panic!("Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB")` when under 24 | a 12 GB card reads 16, a 16 GB card 20: both refused before any allocation | | no |
| The core element threshold | `builder.rs` 41 to 48: `ELEMENT_THRESHOLD` = 2^28 + 2^27 = 402,653,184 elements (`opts.rs` 12) on a card reading over 30 (a 32 GB card reads 36); minus 2^26 + 2^25 + 2^24 = 285,212,672 on a card reading 24 to 30 (a 24 GB card). The environment's `ELEMENT_THRESHOLD` is read at `opts.rs` 129 and then OVERWRITTEN at `builder.rs` 48, which is why the sweep's `ELEMENT_THRESHOLD` rows changed nothing; `HEIGHT_THRESHOLD` survives (it is not overwritten), which is why the 2^25 + 2^20 row did | sets the next two terms | | no |
| The core trace area, one per shard in flight | `builder.rs` 70 to 71: `num_elts = element_threshold + 2^21` (`CORE_LOG_STACKING_HEIGHT` 21, `crates/prover/src/components.rs` 16) = 404,750,336; allocated on the device at `sp1-gpu/crates/jagged_tracegen/src/lib.rs` 484 to 500 (`allocate_and_initialize_traces`: `max_trace_size` felts + `max_trace_size / 2` u32 column index + 2^14 u32) | 6 bytes an element: **2.26 GiB** for the full threshold, 1.59 GiB for the 24 GB threshold | yes, in full, whatever the shard holds | no |
| The program's preprocessed traces (the proving key) | `sp1-gpu/crates/shard_prover/src/setup.rs` 47 to 58 and 107: the same `allocate_and_initialize_traces(max_trace_size)` at `Setup`, kept in the key cache for the connection's life (`server.rs` 139 to 143) | another **2.26 GiB**, held from `Setup` on | yes | no |
| The pinned host trace buffers | `sp1-gpu/crates/prover_components/src/components.rs` 99 to 103: 4 `PinnedBuffer` of `max_trace_size` felts per prover (core 4 x 1.51 GiB, recursion 4 x 0.5 GiB, shrink 4 x 0.125 GiB, wrap 4 x 0.32 GiB) | 9.8 GiB of pinned host RAM, not device memory (the WSL2 working set the bench saw) | no | no |
| The recursion trace area | `builder.rs` 15 and 95: `RECURSION_TRACE_ALLOCATION` = 2^27 elements, one per recursion tracegen (the recursion program's key at setup and its shard at prove) | 0.75 GiB each | yes | no |
| The shrink and wrap provers | `builder.rs` 16, 19, 117 to 127: 2^25 and 85,376,340 elements, built at `Setup` for every proof mode, used only by the Groth16 and PLONK path | host pinned at build; device only when a wrap runs (never, for a compressed proof) | no | no |
| The codewords (LDE) and the Merkle trees | `sp1-gpu/crates/basefold/src/fri.rs` 92 to 97: every stacked column of 2^21 rows encoded to 2^(21 + 1) rows (`log_blowup` 1); kept until the query phase unless `drop_ldes` (`builder.rs` 52: only on a 24 GB card with `FULL_SIZE_SHARDS`); the preprocessed codewords live in the key | 2 x the padded trace, so up to 2 x the term above | yes | yes, with the padded trace |
| The LogUp GKR layers | `builder.rs` 51: `recompute_gkr_trace = false`, so the first layer stays materialised (`sp1-gpu/crates/logup_gkr/src/tracegen.rs` 169 to 224) | of the order of the interaction count | yes | yes |
| The allocator | `sp1-gpu/crates/cuda/src/task.rs` 152 and 196: the device's default `cudaMallocAsync` pool with its release threshold at `u64::MAX`, so nothing freed is ever returned to the driver: `nvidia-smi` reads the high-water mark of everything live at once | | | |
So at zero cycles the server already holds the proving key's 2.26 GiB, the shard's 2.26 GiB (both allocated at the
threshold, not at the shard's rows), their codewords and trees, and the recursion program's key and traces (2 x
0.75 GiB and their codewords): the 13.9 GB floor. The shard's own content only adds to the codewords, the GKR
layers and the working buffers, which is the 13.9 to 20.4 GB step from 0.3 M to 4.7 M cycles, and the flat 28.3 GB
from 20 M cycles is the threshold's padded area reached. The witness (5 to 22 KB) never appears.
## What the patch does (`proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, three files)
1. `builder.rs`: the panic is gone; the card's memory (or `SP1_GPU_MEMORY_BUDGET_GB`) picks the element threshold
from a tier table (`element_threshold_for_budget`: over 30 as read, the full 402.6 M; 24 to 30, upstream's 24 GB
figure; 18 to 24 (a 16 GB card), 2^27 + 2^26 = 201.3 M; under 18 (a 12 GB card), 2^27 = 134.2 M);
`SP1_GPU_ELEMENT_THRESHOLD` sets it directly and `SP1_GPU_RECURSION_TRACE_ALLOCATION` the recursion buffer.
The chosen numbers are printed as a `FLOOR opts` line. Every other option is as upstream.
2. `jagged_tracegen/src/lib.rs`: with `SP1_GPU_FLOOR_LOG` set, every trace allocation prints its capacity and,
after the shard's traces are in, the elements actually used and the device memory in use.
3. `server.rs`: a `FLOOR memory` line (device used, free, total) after `Setup` and after every proof, with the
proof's time.
Nothing in the proof changes: the element threshold only moves where the executor splits shards, exactly what
upstream's own 24 GB tier does with the same verifier and the same keys; the recursion program, the verifying key
and the pinned guest ids are untouched. The unpatched verifier (the pv1 host's SDK) is the one that verifies every
measured proof below.
## The build (PC 2, WSL2 Ubuntu-24.04, job `floor-toolchain-1` then the build job)
Toolchain found 22:10Z (job `floor-toolchain-1`, 4 s): nvcc 12.8 at `/usr/local/cuda-12.8`, cmake 3.28.3, gcc 13.3,
clang 18, protoc 3.21.12, cargo 1.99.0, no Go. The release workflow installs Go for the server's `native-gnark`
feature (the Groth16 and PLONK wrap through gnark), which a compressed proof never runs, so the build drops that
feature from `sp1-gpu/crates/server/Cargo.toml` and nothing else. `CUDA_ARCHS=86,89,120` (consequences reviewer
C26): the 12 GB tier is sm_86 (RTX 3060) and sm_89 (RTX 4070), the 16 GB tier sm_89 and sm_120 (RTX 5080), PC 2's
5090 is sm_120; the stock server lists sm_80, 86, 89, 90, 100 and 120, which a shipped build repeats. The recipe:
`tools/prover-floor/pc2-build-server.ps1` (generated by `make-build-playbook.sh` from the patch, so the two cannot
drift): clone the tag, `git apply` the patch, `touch` the three files, `cargo build --release --bin sp1-gpu-server`
niced with 8 jobs into `/opt/igneum-floor/target`, the binary copied to `/opt/igneum-floor/home/.sp1/bin/` (the SDK
spawns the server it finds under `$HOME/.sp1/bin`, so `HOME=/opt/igneum-floor/home` selects it and the live
`/root/.sp1/bin/sp1-gpu-server` stays as it is).
What a measurement on PC 2 can and cannot say (C26). The server's allocation pattern is deterministic in the
budget it is given, so a run with `SP1_GPU_MEMORY_BUDGET_GB=12` on the 5090 shows the peak a 12 GB card's build
would ask for; it does not show that a 3060 proves it in time, nor what the card's display and driver hold. The
public line keeps "24 GB" until the on-order 12 GB card runs the same fixture. Every row names the arch list and
the card.
What shipping it costs (C26). A patched server means the project signs and distributes its own build of SP1's
prover: the WSL2 package, the DMG's prover inputs, the K1-signed inputs and `evidence.md` carry it, and every SP1
upgrade repeats the clone, patch, build and measurement. The verifying key and the pinned guest ids do not move
(the patch changes buffer sizes and the shard split, not the circuits), which the `verify-segment` and `--mode
compressed` VERIFIED lines of the unpatched host show on every row below. The packaging path is a row for the
proving plan before 0.3.12, not this branch.
## Step 4 contingency, read not measured: RISC Zero's CUDA prover and its memory per segment
If SP1 could not be brought under 11 GB, the alternative's floor is read from its operators' documentation (not
measured here; a PC 2 run would be the measurement): Boundless' prover guide
(https://docs.boundless.network/provers/performance-optimization) sets the segment size cap by VRAM as 8 GB:
po2 19, 16 GB: po2 20, 20 GB: po2 21, 40 GB: po2 22, with measured peaks po2 20: 13,835 MiB, po2 21: 22,905 MiB,
po2 22: 41,089 MiB; RISC Zero's PR 3761 adds `low_vram` and `pinned_witgen` to fit po2 22 on a 24 GB 4090. So
RISC Zero proves a 2^19-cycle segment inside 8 GB and a 2^20 one inside 16 GB, and a shard of 4.7 M cycles is
9 segments at po2 19 plus lift and join steps (times not on the page). Adopting it would cost a second guest (the
chain rule in the RISC Zero zkVM), a second pinned program id, a second verifier in the node and no shared
aggregation between the two formats: `docs/analysis/amd-proving.md` and the proving plan carry that row already.
### The build, as it ran (job `floor-build-3`, 22:28:24 to 22:32:29Z)
Three runs: `floor-build-1` (22:17Z) and `floor-build-2` (22:24Z) failed in 2 to 4 minutes on
`crates/recursion/gnark-ffi/build.rs:70`, "Failed to build Go library: NotFound" (no `go` on PC 2; the first run's
playbook lost its own log, a bug fixed before the second). `floor-build-3` fetched go1.27.1 (tarball sha256
`63d339f0da5ab53635a56f2490a7984dfe12dfcff22ad749f63edaf590168445`, checked before unpacking under
`/opt/igneum-floor/go`, on the job's PATH only) and built in **240 s** (46 crates on the warm target of run 2, 8
niced jobs, 16 cores). The binary: `/opt/igneum-floor/bin/sp1-gpu-server`, **166,768,224 bytes, sha256
`5568108bf7fb9b0e525d8a08926b7046e51136ffaea53f0ca858631d0e938878`**, `--version` 6.8.1, `cuobjdump --list-elf`
sm_86, sm_89, sm_120 (the stock 251,306,680-byte server lists sm_80, 86, 89, 90, 100, 120 and compute_120 PTX).
The live `/root/.sp1/bin/sp1-gpu-server` (c2642ad1...) was never touched; the miners mined throughout.
## Sweep 1 (job `floor-sweep-1`, 22:34:56 to 22:37:55Z): the shard term gone, a second floor found
PC 2's RTX 5090 (32,607 MiB, idle 1,755 MiB with the miners stopped and the live prover off), the patched server
`5568108b...` (sm_86, sm_89, sm_120; the build above), the unpatched pv1 host `dae6b006...` as client and
verifier, one `--mode compressed --shard 0` per point, every server killed and its socket unlinked around every
point, peak = `nvidia-smi memory.used` at 1 s (the idle 1,755 MiB inside it), time = the compressed proof.
Every proof VERIFIED (1,272,897 bytes, verify 0.037 to 0.040 s), so the unpatched verifier accepts every proof
of the patched server.
| Config (environment to the patched server) | Fixture | Cycles | Peak MiB | Prove s | Verified |
|---|---|---|---|---|---|
| control: `SP1_GPU_MEMORY_BUDGET_GB=32` (upstream's sizes) | empty live shard (block 83616) | 280,706 | 13,892 | 2.2 | yes |
| control | v1 shard (fees-v1-shards2 shard 0) | 4,717,439 | 20,516 | 4.2 | yes |
| 12 GB tier: budget 12 (threshold 2^27) | empty | 280,706 | 12,740 | 2.4 | yes |
| 12 GB tier | v1 shard | 4.7 M | 15,396 | 4.1 | yes |
| 12 GB tier + `SP1_WORKER_NORMALIZE_PROGRAM_CACHE_SIZE=1` | v1 shard | 4.7 M | 15,428 | 4.0 | yes |
| 16 GB tier: budget 16 (2^27 + 2^26) | v1 shard | 4.7 M | 18,628 | 3.8 | yes |
| `SP1_GPU_ELEMENT_THRESHOLD=67108864` (2^26) | empty | 280,706 | 12,772 | 3.1 | yes |
| 2^26 | v1 shard (split into 4 core shards) | 4.7 M | **12,708** | 5.3 | yes |
| `SP1_GPU_ELEMENT_THRESHOLD=33554432` (2^25) | v1 shard | 4.7 M | 12,836 | 8.5 | yes |
Reading. The control reproduces the proving agent's curve (13.9 and 20.4 GB), so the patched server behaves as
the stock one at the stock sizes. The shard's term follows the threshold as the model says (20.5 GB at 402 M
elements, 15.4 at 134 M, 12.7 at 67 M), and then stops: 2^26 and 2^25 both sit at 12.7 to 12.8 GB for the empty
shard and the v1 shard alike. The `FLOOR memory after setup` line names the rest: **9,703 MiB in use before the
first shard** (2^26; 11,623 at the stock sizes), and the `FLOOR tracegen alloc` lines at Setup are five
allocations of 134,217,728 elements (the recursion keys, 0.75 GB each, each using 90,177,536 elements: 35.6 M
preprocessed and 54.5 M main at prove time), one of 33,554,432 (the shrink key, 0.19 GB) and one core key at the
threshold. The `NORMALIZE_PROGRAM_CACHE_SIZE` knob does not reach them (they are keys built at `Setup`, not the
program LRU). The time cost of the split: the v1 shard at 2^26 is 4 core shards and 5.3 s against 4.2 s (1.26x);
at 2^27 it is 4.1 s with no split.
So after sweep 1 the binding term is the Setup-time keys allocated at full capacity, and patch v2 sizes every
trace buffer (keys and shards) to its padded need: `padded_trace_elements` in `jagged_tracegen/src/lib.rs`
(each phase pads to the next multiple of 2^21 rows, `generate_jagged_traces`'s "final padding"), applied in
`setup_tracegen` and `full_tracegen`, one stacking height of slack, `SP1_GPU_FLOOR_EXACT=0` restoring upstream.

View file

@ -0,0 +1,411 @@
# Proving methods: why the prover needs 14 GB, what else exists, and how a 12 GB card gets to prove
5 October 2026, from Josh at 22:05 UTC: "if this doesn't enable 12 GB cards, then do a full deep research task on proving
and see if there are different methods." "This" is the prover-floor agent's patch of SP1's GPU server (branch
`prover-floor`), running tonight. This document is research and reading, not measurement: every number of ours is from
`docs/bench-log.md` with its entry named; every claim about another system cites its repository file, its documentation
page or its paper, or is labelled approximate. Status words follow `docs/spec/00-overview.md` 0.2. Day estimates follow
Josh's rule of 3 October 2026: hours of agent time, never weeks.
The facts this starts from (bench-log, "proving v1", 5 October 2026; `docs/plans/proving-v1.md`; `docs/analysis/amd-proving.md`):
| Fact | Number |
|---|---|
| SP1 6.8.1's GPU server, an empty shard (280,706 cycles), the card to itself | 13,874 MiB peak, 2.2 s compressed |
| The adopted v1 shard (`S_p` 30,000 pgas, 4,717,439 cycles) | 20,434 MiB, 4.3 s; 22,210 MiB and 13.2 s beside the miner |
| The prototype shard (6.75 M pgas, 60.4 M cycles) | 28,307 MiB, 10.8 s; flat at 28.3 GB from 20 M cycles up |
| Aggregation, chained, per block, on a mining 5090 | 9.6 to 9.7 s; 2.2 to 2.5 s with the card to itself |
| The CPU path (PC 1, 16 cores) | 282 s a shard whatever its size, 29.5 to 30.5 GB RSS |
| AMD and Apple GPUs | no zkVM proves on AMD; RISC Zero has a Metal prover, SP1 does not |
| The promise | `site/litepaper.html`: "Target: shard size will be set so a 12 GB card proves one shard in about 20 seconds"; the design goal is every block proven within about a minute by the miners' own cards |
## 1. The memory anatomy of a STARK-based zkVM prover, and why the floor is where it is
### 1.1 What SP1 6.8.1 is
SP1 6.x is not the univariate FRI STARK of the earlier SP1 releases (Succinct calls Hypercube the "first zkVM built entirely on a multilinear polynomial-based proof system", blog.succinct.xyz, sp1-hypercube, 20 May 2025). The crates it pulls say what it is: `slop-multilinear`, `slop-sumcheck`,
`slop-jagged`, `slop-stacked`, `slop-basefold`, `slop-whir` (the `~/.cargo/registry` of this Mac; `proving/igneum-prove/Cargo.toml`
pins `sp1-sdk = "=6.8.1"`). The architecture Succinct calls Hypercube: the execution trace is a set of multilinear
polynomials over the 31-bit KoalaBear field (`sp1-hypercube-6.8.1/src/verifier/config.rs`: `SP1BasefoldConfig =
Poseidon2KoalaBear16BasefoldConfig`), the constraints are checked by a zerocheck sumcheck and the lookups by a LogUp GKR
(`sp1-gpu/crates/zerocheck`, `sp1-gpu/crates/logup_gkr`), and the polynomial commitment is "jagged": every table's
columns, whatever their heights, are concatenated into one long vector, stacked into rows of height `2^log_stacking_height`
and committed with BaseFold, a FRI-like folding over a Reed-Solomon code (`sp1-gpu/crates/basefold/src/fri.rs`,
`slop-basefold-6.8.1/src/verifier.rs`). The proof system parameters, from `sp1-primitives-6.8.1/src/fri_params.rs` and
`sp1-prover-6.8.1/src/components.rs`:
| Parameter | Value | Where |
|---|---|---|
| Field | KoalaBear, 31 bits, 4 bytes an element; extension degree 4 (16 bytes) | `sp1-primitives` |
| Core stage: Reed-Solomon blowup | `CORE_LOG_BLOWUP = 2`, so the codeword is 4x the data | `fri_params.rs:5` |
| Core stage: stacking height, maximum rows per table | `CORE_LOG_STACKING_HEIGHT = 21`, `CORE_MAX_LOG_ROW_COUNT = 22` | `components.rs:16,17` |
| Core shard limits (the executor's cut) | `MAX_SHARD_SIZE = 2^24` cycles, `ELEMENT_THRESHOLD = 2^28 + 2^27 = 402,653,184` trace elements, `HEIGHT_THRESHOLD = 2^22` rows | `sp1-core-executor-6.8.1/src/opts.rs:9-12` |
| Recursion (compress) stage | blowup 2 (`RECURSION_LOG_BLOWUP = 2`), stacking height 20, max rows 2^21 | `fri_params.rs:6`, `sp1-verifier-6.8.1/src/compressed/config.rs:1,2` |
| Shrink and wrap stages | blowup 3 (8x), 22 bits of grinding, stacking 18 and 21 | `fri_params.rs:17,18,7`, `components.rs:37-40` |
| Recursion arity | 4 proofs per compose step (`DEFAULT_MAX_COMPOSE_ARITY = 4`, `DEFAULT_MAX_REDUCE_ARITY = 4`) | `sp1-prover-6.8.1/src/worker/config.rs:183,193` |
| Workers | 4 core workers, 8 recursion prover workers, 4 recursion executors, 4 deferred workers, buffers of 4 to 8 | `worker/config.rs:188-205` |
The stages a shard goes through (`sp1-prover-6.8.1/src/worker/controller/*.rs`): execute (the RISC-V executor cuts the
run into core shards at the thresholds above); core (one jagged-PCS proof per core shard, on the GPU); normalize and
compose (each core proof is verified inside a recursion program, then proofs are folded 4 at a time until one remains,
the "compressed" proof, 1,272,897 bytes for every shard we have proven, bench-log 4 and 5 October); deferred (what the
aggregator uses: `verify_sp1_proof` inside a guest, `proving/igneum-prove/aggregator/src/main.rs`); shrink and wrap
(to a BN254 STARK, then Groth16 or Plonk; not run here, ledger P3).
### 1.2 The terms, and which scale with the shard
Every STARK-family prover holds these buffers on the device at its peak, in some order and with some overlap. The
sizes below are from the constants of 1.1 and the allocation code of `sp1-gpu`; where a buffer's size is the actual
trace rather than the maximum, the row says so.
| Term | What it is | Size rule | SP1 6.8.1 on a 32 GB card | Scales with the shard? |
|---|---|---|---|---|
| Main trace | the witness: one element per cell of every table the shard touched | `cells x 4 bytes`, where cells = sum over tables of rows x columns; the executor cuts a new core shard at 402,653,184 cells | the device buffer is allocated at the MAXIMUM, not the actual trace: `allocate_and_initialize_traces` takes `max_trace_size` and allocates `max_trace_size` felts plus `max_trace_size / 2` u32 of index (`sp1-gpu/crates/jagged_tracegen/src/lib.rs:484-503`), 6 bytes a cell; the core prover's `max_trace_size` is `element_threshold + 2^21` (`prover_components/src/builder.rs:70-71`): **2.26 GiB** on a card over 30 GB, 1.61 GiB on a 24 GB card (the threshold drops by 2^26 + 2^25 + 2^24 when memory is 30 GB or under, `builder.rs:41-45`) | no: fixed at the maximum shard, whatever the trace |
| Preprocessed trace | the program's own tables (the ELF as a `Program` AIR, the byte and range tables) | program size x its columns plus 2 x 2^16-class tables | small for a 2.8 MB guest ELF (`elf/manifest.json`); not isolated | with the guest, not the shard |
| Codeword (the LDE) | the stacked polynomial encoded at rate 1/4 for BaseFold | `stacked cells x 4 (blowup) x 4 bytes`; the stacked length is the actual cell count padded to a multiple of 2^21 | 4.7 M cycles: approximate, the actual trace; 60 M cycles: the shard is 7 to 15 core shards of up to 402 M cells, each encoded to 6 GiB at the blowup, one or more in flight | yes, up to the core-shard cap; past the cap the shard count grows and the per-shard term stays |
| Merkle commitment | Poseidon2 hashes of the codeword rows | `rows x 8 elements x 4 bytes x 2`, rows = 2^21 x blowup | about 0.5 GiB at full stacking, approximate | with the stacked rows |
| Zerocheck and GKR | the constraint sumcheck over the extension field, and the LogUp GKR layers | extension elements are 16 bytes; the sumcheck holds a folded copy of the trace in the extension field, which is 4x the base trace at the first round and halves each round | up to about 4x the live trace in the first round, approximate (`sp1-gpu/crates/zerocheck/src/primitives.rs:174,287`: `Buffer<Ext>` of `new_total_length`) | yes |
| Recursion traces | the normalize and compose programs' own traces, verifying core proofs | fixed-shape programs: `RECURSION_TRACE_ALLOCATION = 2^27` cells (`builder.rs:15`), allocated at 6 bytes a cell: **0.75 GiB** per recursion prove, at blowup 4 a 2 GiB codeword plus its own zerocheck | fixed per recursion step; the number of steps is log4 of the core-shard count | no (per step) |
| Shrink and wrap traces | the two last stages, not run by us | 2^25 and 85,376,340 cells (`builder.rs:16,19`): 0.19 and 0.48 GiB | only when wrapping | no |
| Proving keys and program cache | the recursion programs (`vk_map.bin`, the normalize cache of 5 programs) and the shard program's setup | `DEFAULT_NORMALIZE_PROGRAM_CACHE_SIZE = 5` (`worker/config.rs:192`); the key setup took 14.6 s on the 5090 (bench-log 4 October) | not isolated | no |
| Pinned host buffers | the staging copies on the PC side | 4 core workers x `max_trace_size` x 4 bytes = **6.0 GiB** of pinned RAM, plus 4 x 0.5 GiB for recursion (`prover_components/src/components.rs:99-103`, `builder.rs:76,95`) | this is the 7.9 GB WSL2 working set measured on 5 October | no |
| The allocator | CUDA's default memory pool with its release threshold set to `u64::MAX` (`sp1-gpu/crates/cuda/src/task.rs:152,190-199`): freed blocks are never returned to the driver | `nvidia-smi` therefore reports the high-water mark of everything above, and it stays until the server exits | this is why the memory curve is flat between shards of different size | no |
Two facts from this table explain the measurements:
1. **The server refuses small cards by code.** `local_gpu_opts()` reads the card's total memory, adds 4 GB, and panics
under 24: `"Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"` (`sp1-gpu/crates/prover_components/src/builder.rs:35-38`).
A 16 GB card (16 + 4 = 20) and a 12 GB card (16) never start; a 20 GB card is the smallest that does. The 13.9 GB
floor measured on the 5090 is therefore not the whole story for a 12 GB card: on this build the card is refused
before any buffer is allocated. Any route through SP1's GPU server starts by removing this line.
2. **The environment knobs do not reach the floor** because the same function overwrites `element_threshold` with the
compile-time constant (`builder.rs:41-48`); only `HEIGHT_THRESHOLD` passes through, which is why the sweep's
`ELEMENT_THRESHOLD 2^26` rows changed nothing and `HEIGHT_THRESHOLD 2^20` took 5.4 GB off the 60 M-cycle shard
(bench-log, "the 12 GB requirement", 5 October 2026). The worker counts only slow the proof (11.4 s to 20.8 s)
because the device buffers are sized by `max_trace_size`, not by the worker count.
### 1.3 Why the floor is 13.9 GB for an empty shard
With the release threshold at `u64::MAX`, the peak is the high-water mark over the whole pipeline. For an empty shard
the core trace is small (280,706 cycles), so the fixed-shape terms dominate: the 2.26 GiB main-trace buffer allocated
at the maximum, the recursion step over a fixed-shape normalize program (a 0.75 GiB trace buffer, its 4x codeword in
the extension field for the zerocheck, its Merkle tree), the proving-key and program caches, and the deferred and
compose machinery that a compressed proof always runs once. The decomposition of the 13.9 GB into those terms is
approximate until the prover-floor agent's profile lands (branch `prover-floor`, tonight): the figure that is not
approximate is that none of it is the witness (5 to 22 KB a shard) and none of it is the shard's cycles (the same
13.9 GB at 280 k cycles and 556 k cycles, bench-log "the S_p curve").
The step from 13.9 GB (empty) to 20.4 GB (4.7 M cycles) is the live trace: the 4.7 M-cycle shard is one core shard
(its trace area is under 402 M cells, so it was not split; approximate from the memory curve, the cell count is not
logged by the host), and its codeword, zerocheck and GKR buffers are sized by its actual cells. The step from 20.4 GB
to 28.3 GB (20 M cycles and up) is the second and later core shards in flight at once: 4 core workers with a buffer of
4 (`worker/config.rs:188,189`) let several core shards' codewords exist at the same time; past 20 M cycles the pipeline
is full and the peak is flat, which is what the curve shows (28,371 MiB at 20 M cycles, 28,307 at 40 M and 60 M).
### 1.4 The theoretical floor for our guest at the adopted shard
If every buffer were sized to the shard rather than to the maximum, the adopted v1 shard (4.7 M cycles) would need,
approximate, from the rules of 1.2:
| Term | Rule | Approximate bytes |
|---|---|---|
| Main trace, actual | 4.7 M cycles x about 60 cells a cycle (the `Add` and `Addi` tables cost 33 and 30 columns a row, a memory access adds 20 and a global interaction 241: `sp1-core-executor-6.8.1/src/artifacts/rv64im_costs.json`) | about 280 M cells, 1.1 GB |
| Codeword at blowup 4 | 4x | 4.5 GB |
| Zerocheck first round in the extension field | 4x base, halving each round | 4.5 GB at the peak round, falling |
| Merkle tree | rows x 32 bytes x 2 | 0.3 GB |
| Recursion step, fixed | 2^27 cells x 6 bytes plus its 4x codeword and extension copies | 2 to 3 GB, approximate |
| Keys and caches | | under 1 GB, approximate |
| Peak, if the core stage and the recursion stage do not overlap and the pool releases | | **about 10 to 11 GB**; about 6 GB if the blowup-4 codeword is replaced by a rate the sumcheck does not need (see 2.4) |
So the adopted shard is, on paper, a 12 GB card's shard with the server re-sized and nothing else changed, and it is a
12 GB card's shard with 4 GB to spare if the shard is halved (`S_p` 15,000 pgas, 2.4 M cycles: the planner cuts at
transaction boundaries to any budget, `core/src/plan.rs`, and the fee switch of 5 October already moved `S_p` once).
What the paper figure does not say is the time: a smaller card proves slower, and 60 s with the miner running is the
bound (section 3). The prover-floor agent is measuring the real figure; this section says what it should find and why.
### 1.5 RISC Zero's anatomy, for comparison
RISC Zero is the FRI STARK the textbooks describe, and its constants make the same table easy to read
(`~/.cargo/registry`, `risc0-zkp-3.0.4/src/lib.rs`, `risc0-circuit-rv32im-4.0.4/src/zirgen/defs.rs.inc`):
| Term | Value | Where |
|---|---|---|
| Field | BabyBear, 31 bits; extension degree 4 | `risc0-core` |
| Segment size | `DEFAULT_SEGMENT_LIMIT_PO2 = 20` (1,048,576 cycles), `MIN_CYCLES_PO2 = 13`, `MAX_CYCLES_PO2 = 24`; `DEFAULT_MAX_PO2 = 22` for the verifier | `risc0-circuit-rv32im-4.0.4/src/execute/mod.rs:39`, `risc0-zkp-3.0.4/src/lib.rs:35-38`, `risc0-zkvm-3.0.4/src/receipt.rs:884` |
| Trace width | data 211 + accum 103 + code 1 = 315 columns; globals 90, mix 36 | `defs.rs.inc:7-11` |
| Blowup | `INV_RATE = 4`; 50 queries; FRI fold 16 | `risc0-zkp-3.0.4/src/lib.rs:41-51` |
| Recursion | lift, join and resolve programs at `RECURSION_PO2 = 18` rows | `risc0-zkvm-3.0.4/src/host/recursion/prove/mod.rs:58` |
| The GPU buffers | `check + ctrl + data + accum + mix + out` elements x 4 bytes, printed by the CUDA HAL at `eval_check` | `risc0-circuit-rv32im-4.0.4/src/prove/hal/cuda.rs:181-200` |
From those constants the trace of a segment is `2^po2 x 315 x 4` bytes and its LDE 4x that, so, approximate: po2 18 is
0.3 GB of trace and 1.5 GB with the LDE, po2 19 is 3.1 GB, po2 20 is 6.2 GB, po2 21 is 12.3 GB, before the check
polynomial, the extension-field accumulators and the Merkle trees. Two things follow. A RISC Zero segment at the
default 2^20 is in the same class as one SP1 core shard, not smaller. And a RISC Zero segment at 2^18 or 2^19 is a
2 to 4 GB object: the only reason a 12 GB card could not prove one is the fixed overhead of the recursion circuits
(2^18 rows each) and the allocator, which is the measurement the prover-floor agent takes on PC 2 if SP1 cannot go
under 11 GB. Section 2.2 carries the documented numbers.
### 1.6 Which terms the shard size can move, and which it cannot
| Lever | Moves | Does not move |
|---|---|---|
| Our `S_p` (pgas per shard) | the live trace, the codeword, the zerocheck: everything in 1.2 marked "yes" | the maximum-sized buffers, the recursion step, the keys, the pool |
| SP1's `HEIGHT_THRESHOLD` (the one knob the server honours) | the rows per table in one core shard, so the live buffers | the fixed terms (measured: 13,861 MiB on the empty shard with every knob at its minimum) |
| A server patch: size `max_trace_size` to the shard, release the pool, one core worker | the 2.26 GiB buffer, the high-water mark, the in-flight count | the recursion step's fixed shape and the key caches |
| A different proof system | the blowup (sumcheck-only and linear-code systems have none, 2.4), the recursion shape | the trace itself: a RISC-V cycle costs tens of cells in every zkVM |
## 2. Every current proving route
Read 5 October 2026, 22:10 to 23:00 UTC, by four research agents and this one; every cell names its page or file.
"Not documented" means the project publishes no figure, which for a memory floor is itself the finding.
### 2.1 The zkVMs with a GPU prover
| Prover | Proof system, field, chunk | GPU support and the documented minimum memory | Throughput, on what | Verification of the recursive proof; wrapper | Licence | State, October 2026 |
|---|---|---|---|---|---|---|
| **SP1 6.8.1** (ours) | Hypercube: multilinear, jagged PCS, BaseFold, LogUp GKR; KoalaBear; core shards of up to 2^24 cycles and 402 M cells (section 1) | CUDA only. Docs: "24GB or more VRAM", compute capability 8.0+, Linux x86_64 (docs.succinct.xyz, hardware-acceleration page). Code: panic under 20 GB physical (`builder.rs:35-38`). Measured here: 13.9 GB floor, 20.4 GB at the adopted shard, 28.3 GB at the prototype shard. Issue #2950: two clients on a 48 GB L40S hold 41 to 43 GB; a single 6 GiB tensor allocation failed | 4.3 s for the adopted shard, 10.8 s for the prototype one on a 5090 (bench-log); Succinct: 99.7% of Ethereum blocks under 12 s on 16 x RTX 5090 (blog.succinct.xyz, 18 Nov 2025) | compressed proof 1,272,897 bytes, verified in 0.032 to 0.040 s here (`--mode verify-segment`); Groth16 about 260 bytes and about 270 k gas, Plonk about 868 bytes and 300 k gas (docs, proof-types page); the Groth16 wrap needs about 14 GB of host RAM, Plonk about 60 GB (hardware-requirements page) | Apache-2.0 or MIT for the repository including `sp1-gpu/` (`LICENSE-APACHE`, `LICENSE-MIT` at the root; no separate licence under `sp1-gpu/`); `sp1-cluster` is Business Source 1.1 | v6.8.1 of 24 Sep 2026 is the latest tag; mainnet for Ethereum proving since 19 Feb 2026 (blog); AMD port PR #2668 closed unmerged 20 Mar 2026; no Metal, Vulkan or WebGPU |
| **RISC Zero 3.0.x** | FRI STARK (DEEP-ALI), BabyBear, Poseidon2, blowup 4, 50 queries; segments of `2^po2` cycles, default po2 20, allowed 13 to 24 (section 1.5); lift, join, resolve recursion at 2^18 rows; keccak as a separate circuit | CUDA and **Metal** (`risc0/sys/kernels/zkp/metal/`; on Apple silicon the Metal path is on automatically, `risc0/zkvm/build.rs`). Documented memory per segment: Bento design page, 1 M cycles 9 to 10 GB, 2 M 17 to 18 GB, 4 M 32 to 34 GB; Boundless performance page, the largest `SEGMENT_SIZE` per card: 8 GB card po2 19, 16 GB po2 20, 20 GB po2 21, 40 GB po2 22; docs: "less than 10 GB available: change the segment size limit" (dev.risczero.com, local proving); `env.rs:190-192`: "lowering this value by 1 will cut memory consumption by about half". PR #3761 (June 2026): po2 22 did not fit a 24 GB 4090 until the `low_vram` buffer reuse | 4090: 808 kHz at po2 21, 1,207 kHz at po2 22 with PR #3761 (end to end to a succinct receipt); Apple M2 Pro about 14 kHz on the 2023 datasheet, approximate (the page was unreachable tonight); real-time Ethereum on about 160 x 4090 (blog, approximate) | succinct receipt 222,668 bytes, constant; about 100 ms to verify, approximate (`gsr-stark-verifier` PR #5, mirrored docs); Groth16 seal 256 bytes, about 200 to 300 k gas, approximate; the Groth16 wrapper is x86 only, not on Apple silicon (docs) | Apache-2.0 or MIT, CUDA and Metal kernels included (`risc0/sys/kernels/zkp/cuda/eltwise.cu:1-13`); Bento is BSL 1.1 with a change date already passed | v3.0.6 of 17 Jul 2026 on the maintained line; `main` is 5.0.0 with no release body; `RISC0_PROVER=actor` multi-GPU scheduler experimental since 3.0.1 (`r0vm/src/actors/factory.rs:183-195` carries measured per-po2 memory tokens: po2 18 = 8, 19 = 10, 20 = 15, 21 = 24; lift and join = 3) |
| **Airbender** (Matter Labs) | DEEP STARK, FRI, Mersenne31; chunks of 2^22 cycles; Boojum then FFLONK wrap (docs.zksync.io, airbender page) | CUDA only. "any GPU with 22GB RAM" for production (zksync.io/airbender); the code has memory presets `GiB21` (24 GB cards) and `GiB30`, raised from 29 because Ethereum blocks failed to allocate at 29 GiB (PR #448, `gpu/execution_prover/src/prover/config.rs`); the final SNARK is CPU with about 150 GB of RAM (`docs/gpu.md`) | one H100: 21.8 MHz base layer, 8.5 MHz end to end, about 35 s an Ethereum block (June 2025 post); ethproofs.org today: 4 x 5090 2.3 s average | FFLONK over BN254 on chain; gas not published | MIT or Apache-2.0 | v0.6.0-rc.2; Veridise audit Feb to Apr 2026; live for ZKsync Atlas chains |
| **ZisK** (Polygon spin-out) | eSTARK over Goldilocks (pil2-stark), Poseidon2, approximate; main instance 2^22 to 2^23 rows, chunks of up to 2^22 steps (PR #1238) | CUDA only, CUDA 12.9+; **no VRAM floor documented**; workers need about 32 GB of host RAM, the assembly emulator 64 GB (docs, limits and distributed pages); Cysic's Venus fork submits from one RTX 4090 | 4 x 5090: p99 9.62 s on Ethereum blocks (Aug 2026); 24 x 5090 6.56 s average (Nov 2025) | PLONK wrapper verified by Solidity (`zisk-contracts`); 128-bit claimed | Apache-2.0 or MIT | v1.3.1-alpha, 30 Sep 2026, "undergoing security and correctness audits" (README) |
| **OpenVM 2.0** (Axiom) | SWIRL: sumcheck, zerocheck, LogUp GKR, stacked reduction into WHIR; BabyBear; segments by metered trace height (blog.openvm.dev/2.0) | CUDA only; "at least 24GB of VRAM": L40, 4090, L40S, 5090 (blog.openvm.dev/openvm-gpu) | 11.4 MHz on one 5090, 139 MHz on 16; 2.1 preview: 4 x 5090 p99 9.7 s | STARK proof under 300 KB; Halo2-KZG wrapper, 316 k gas; the Halo2 wrap 8.1 s on a 5090 | MIT or Apache-2.0, GPU prover included | v2.0.2 of 14 Aug 2026; zkSecurity audit of SWIRL; Scroll's prover builds on it |
| **Pico** (Brevis) | Plonky3 STARK, KoalaBear default; chunk size a parameter with no documented default | CUDA via `pico-gpu`; **no VRAM figure published**; every run on 32 GB 5090s | Prism 2.1: 16 x 5090 over two machines, 4.87 s average on Ethereum blocks | Groth16 via gnark | core MIT or Apache; **`pico-gpu` is BUSL-1.1** and "not recommended for production" (its README) | v2.1.2, Aug 2026; Sherlock audit |
| **Ziren** (ZKM, MIPS) | Plonky3-class, KoalaBear, LogUp GKR, WHIR | CUDA 12, compute capability 8.6+, "24 GB VRAM or higher"; the GPU prover is a Docker image pinned by digest, source "planned H1 2026" (docs.zkm.io prover page; an independent evaluation of v1.1.4 says the GPU path is not open) | one 5090: 5.9 MHz on a 288 M-cycle block; 4 GPUs 3.1 to 3.3x | compressed proof 603 KiB; Groth16 or PLONK | core MIT or Apache; GPU image licence unspecified | v1.2.7; no public audit cited |
| **Stwo / S-two** (StarkWare) | Circle STARK over Mersenne31; blowup 1 (rate 1/2), 70 queries, 26 bits of grinding | **CPU SIMD first** (AVX2, AVX-512, NEON, WASM); GPU through ICICLE-Stwo (Ingonyama): about 3 GB of trace in GPU memory, out of memory from 2^23 rows; a WebGPU port of the constraint evaluation (zkSecurity blog, April 2025); client-side proving under 1 GB after a spill allocator (third-party PR) | 620 k Poseidon2 a second on an M3 laptop | via a Cairo verifier (proofs of proofs); sizes not published here | Apache-2.0 | live on Starknet mainnet since 3 Nov 2025; no RISC-V guest of its own (Nexus 3.0 is the RISC-V zkVM on it, BUSL-1.1 until 2029) |
| **Jolt** (a16z) | sumcheck and lookups (Lasso lineage, Twist and Shout memory checking); PCS Dory over BN254 by default, or **Akita**, a lattice commitment over a 128-bit prime field (Sep 2026, "Lattice Jolt"); RV64IMAC; no continuations (the book's recursion page is "under construction") | **No CUDA in the public repository** (LayerZero's "Jolt Pro" CUDA port is private); **Metal**: PR #1938 merged 30 Sep 2026 (the `jolt-metal` runtime crate), PR #1733 (the full prover on Metal) still a draft. Memory: "about 200 bytes per cycle" with Akita (a16z substack, Sep 2026); the book: "under 2 GB of memory per million cycles"; a streaming prover bounded to "a few GBs" is planned, not shipped | over 2 M cycles a second on a laptop CPU with Akita, over 10 M with Metal on a Apple laptop (a16z substack, Sep 2026); PR #1733: M5 Max, 2^25 cycles in 19.8 s, 2^27 in 77 s at an 89.4 GiB footprint | proof about 50 KB (Dory) or 65 to 80 KB (Akita); verify sub-second, approximate; on-chain 1.3 to 2 M gas estimated in 2024; no Groth16 wrapper shipped | MIT or Apache-2.0 | `v0.3.0-alpha` is the last tag (1 Oct 2025); README: "not suitable for production use"; no audit |
| **Ceno** (Scroll) | GKR tower prover, BabyBear, WHIR or BaseFold PCS; RV32IM | CUDA, but the real HAL is in a **private** `ceno-gpu` repository (the public one is a mock); no memory numbers | 2 GPUs 1.6x over one (PR #1403, Sep 2026) | via OpenVM recursion to Halo2 | Apache-2.0 | README: "under construction and not suitable for use in production" |
| **Nexus 3.0** | on Stwo (Circle STARK, M31) | no GPU path documented | none published | not published | **BUSL-1.1** until 10 Feb 2029 | last push 6 Jan 2026; folding (Nova family) abandoned June 2025 for the STARK |
| **Valida** (Lita) | Plonky3 STARK | no GPU; a CUDA port "underway" in July 2025 | none current | not published | Apache or MIT | dormant since Sep 2025; documented soundness issues in its own benchmarks page |
| **Powdr** | no longer a zkVM: `powdrVM` archived; powdr is autoprecompiles on OpenVM | OpenVM's | OpenVM's | OpenVM's | MIT or Apache | tooling layer; "DO NOT USE FOR PRODUCTION" |
| **Binius / Binius64** (Irreducible) | binary-field SNARK, BaseFold-style FRI over GF(2^64) words | CPU SIMD only; the FPGA work was dropped 9 Sep 2025 ("FPGAs underperformed GPUs"); no GPU | ECDSA aggregation about 5x over SP1 and R0VM on L40S GPUs, on CPU (the page carries methodology corrections) | hash-based; no recursion shipped | Apache-2.0 or MIT | **the company shut down 12 Nov 2025**; the only zkVM on it (PetraVM) is archived |
| 2026 entrants | Cysic Venus (a ZisK fork with cudaGraph tuning and an FPGA backend, Apache or MIT, "do not use in production"); Zilkworm (Erigon's C++ guest on Airbender, 2 x 5090 9.3 s); zkDTVM (evmone guest, 4 x 5090 4.7 s, no public docs); Delphinus zkWasm (Halo2 on BN254, a 4090 minimum plus 58 GB of host RAM); Miden (Goldilocks STARK, client-side, Metal via `miden-gpu`, mainnet alpha planned); Boojum (2023 claim of proving on a 16 GB card, superseded by Airbender) | none states a floor under 24 GB on a GPU | | | | |
The reading of the table. No shipped zkVM documents a GPU floor under 24 GB except RISC Zero, whose memory is a
function of a runtime knob (`segment_limit_po2`) and is published per card size by Boundless. SP1's 24 GB is a line
of code, not a property of the proof system: Airbender, OpenVM and Pico all pad to the card they tune on, and all
three say 24 or 32 GB because their market is Ethereum blocks on 5090 clusters. The real-time race has collapsed to 2
to 4 consumer cards per block (ethproofs.org, 5 October 2026), which is why nobody is tuning for a 12 GB card: the
customer buys 5090s. Igneum's customer is the miner who already owns the card, so Igneum has to do the tuning itself.
### 2.2 The sumcheck and GKR family against FRI STARKs, in memory terms
| Family | What it holds at the peak | Blowup | The GPU figure today | Source |
|---|---|---|---|---|
| FRI STARK (RISC Zero, Airbender, ZisK, Pico, Stwo, SP1 3 and 4) | trace, its Reed-Solomon codeword at the blowup, the Merkle trees, the DEEP quotient in the extension field | 4x (RISC Zero, Airbender), 2x (SP1 3 and 4, approximate), 2x (Stwo at rate 1/2) | RISC Zero: 9 to 10 GB per 1 M cycles (Bento) | section 1.5; Boundless pages |
| Sumcheck with a hash-based PCS (SP1 Hypercube, OpenVM SWIRL, Ceno, Ziren) | the trace as multilinears, the extension-field folded copies of the zerocheck and GKR, and the BaseFold or WHIR codeword of the stacked polynomial (still a Reed-Solomon encoding, at 4x in SP1, section 1.1) | 4x of the stacked data in SP1; WHIR's rate is a parameter | SP1: the fixed 13.9 GB plus about 6.5 GB for a 4.7 M-cycle shard (measured) | section 1 |
| Sumcheck with a curve or lattice PCS (Jolt) | the trace and the one-hot columns; **no codeword at all**: Dory commits by MSM and Akita by lattice hashing, so memory is bytes per cycle with no blowup | none | no GPU figure: 200 bytes a cycle on CPU (Akita), so the adopted 4.7 M-cycle shard is about 0.9 GB of prover RAM, approximate (derived) | a16z substack, Sep 2026; the Jolt book, streaming page |
| GKR (Expander, Ceno) | the circuit witness layer by layer; no codeword for the inner layers | none inside; a PCS for the inputs | Expander: 16 MB per Keccak, approximate | Polyhedra blog (returned 530 tonight) |
| Linear-code PCS (Ligero, Brakedown, Ligerito, Blaze) | one encoded matrix and one Merkle tree; linear time, no FFT | rate 1/2 to 1/4 | no prover memory benchmarks found; Linea's Vortex is the only production use | eprint 2021/1043, 2025/1187, 2024/1609; `linea-monorepo/prover/protocol/compiler/vortex` |
| Binius (binary field) | words of GF(2^64) and a BaseFold FRI | 2x to 4x | none; CPU only; company closed | irreducible.com posts |
The memory law in one line: a FRI or BaseFold prover holds `blowup x trace` plus the trace itself plus extension-field
working copies, so 8 to 12 bytes per cell at the peak; a Jolt-class prover holds the trace and its lookups at about 4
bytes per cell and commits without encoding. The figure that matters for us is not the ratio but the absolute: our
adopted shard is small enough (about 280 M cells, section 1.4) that a FRI-class prover sized to it fits a 12 GB card,
and a Jolt-class one fits a phone. The reason SP1 does not fit today is section 1.3, not the proof system.
### 2.3 Folding schemes
| Scheme | Prover memory per step | The verifier at the end | Field | GPU | Fit for Igneum |
|---|---|---|---|---|---|
| Nova, SuperNova, HyperNova, ProtoStar, Mova; Sonobe as the library | one step's witness plus the running instance: tiny by construction (eprint 2021/370) | an IVC proof of O(F) group elements, compressed by a SNARK: Sonobe's decider is Groth16 over BN254 with KZG, about 11.9 M constraints for a 500 k-constraint step (sonobe.pse.dev, decider page); MicroNova about 2.2 M gas (eprint 2024/2099) | curve cycles (Pasta, BN254 and Grumpkin) | partial: sppark MSM on Pasta, a GPL-3 `cuda-nova` for BN254; Sonobe lists GPU as a plan | **no**: a RISC-V step over a 256-bit curve cycle costs two MSMs per step, the opposite of the hash-based consumer-card design decision (design 5.6), and the verifier changes to pairings |
| Nexus zkVM 1 and 2 | the prover ran "on as little as 1 GB of RAM" (whitepaper, approximate) | a curve SNARK | curve cycle | none | **abandoned by its own author**: Nexus 3.0 (25 June 2025) moved to a Circle STARK, "proofs are smaller, faster to generate" (StarkWare blog) |
| LatticeFold, LatticeFold+, Neo, SuperNeo; Nightstream as the zkVM | one step plus an accumulator, lattice commitments over 64-bit fields | a Spartan-class decider | Goldilocks named; BabyBear and KoalaBear not | none; Nethermind's LatticeFold is a "proof-of-concept prototype" whose benches take 48 h; Nightstream is "research software, not production-ready" with its RV32IM prototype removed | **not before 2027 at the earliest**; the first candidate that folds small-field STARK steps |
| Arc, WARP (hash-based accumulation of Reed-Solomon proximity claims) | small: Merkle openings per step (eprint 2024/1731, 2025/753) | a FRI-style accumulator check | any STARK field | none; no public implementation found | the right primitive on paper for folding RISC-V STARK shards with a hash-based verifier; nothing to adopt |
| Mangrove, Nebula | 390 MB peak at 2^24 gates (Mangrove, eprint 2024/416); pay-per-use steps (Nebula) | curve SNARK | curve cycle | none | research |
What folding would mean for a shard: the shard prover would hold one transaction's step at a time and the memory
floor would vanish; the price is a curve-based decider at the end of every shard (seconds on a CPU, a different
verifier in the node, pairings on the light-client path), and no production code over our field. Today's small-field
zkVMs get their bounded memory from segmenting and recursion (2.4), not from folding. Folding is a watch item, not a
route.
### 2.4 Continuations and segment proving at small sizes
| Prover | The segment knob | What a 2^18 or 2^19 segment costs | Can our shard be cut that way inside the guest? |
|---|---|---|---|
| RISC Zero | `segment_limit_po2`, runtime, 13 to 24 (`env.rs:181-186`); the recursion lifts every segment at 2^18 rows and joins them in a tree | po2 19 is the documented fit for an 8 GB card and po2 20 for a 16 GB card (Boundless); the scheduler's measured tokens put po2 18 at about a third of po2 21 and a lift or join at an eighth (`factory.rs`); the 4.7 M-cycle shard at po2 19 is 9 segments, 9 lifts and 8 joins | yes, with no guest change: the zkVM cuts at the limit on its own; the shard statement is unchanged and one succinct receipt comes out |
| SP1 | `HEIGHT_THRESHOLD` (honoured) and `ELEMENT_THRESHOLD` (overwritten by the server, section 1.2); `SHARD_SIZE` up to 2^24 cycles | the live buffers shrink (22.9 GB against 28.3 GB on the prototype shard at `HEIGHT_THRESHOLD 2^20`, bench-log) and the fixed 13.9 GB does not; the compose tree folds 4 proofs at a time | yes, the same way; but the floor is the server's, so the cut buys nothing until the server is re-sized (route A) |
| Our own planner | `S_p` in pgas, a consensus parameter changed by the fee-switch pattern (`docs/plans/fee-switch-devnet.md`); the cut is at transaction boundaries (`core/src/plan.rs`) | halving `S_p` halves the live trace and doubles the shard count; the aggregator verifies one deferred proof per shard (1.66 M cycles for 4 shards, bench-log 4 October) and the chained aggregation is one per block whatever the count (9.7 s on a mining 5090) | yes, already implemented; a transaction above `S_p` stays one shard and the zkVM's own continuations cover it (spec 7.6 item 1) |
| Jolt | none: monolithic; streaming planned | n/a | no |
### 2.5 Distributed proving across several small cards
| System | How one execution is split | Per-card memory | Several cards on one host | What it means for four 12 GB cards |
|---|---|---|---|---|
| SP1 cluster (`sp1-cluster`, BSL 1.1) | by core shard: `ProveShard`, `RecursionReduce`, `RecursionDeferred` and `ShrinkWrap` tasks go to GPU workers, `CoreExecute` and the Groth16 or Plonk wrap to CPU workers (`crates/prover-types/src/lib.rs:31-41`); artifacts through Redis and S3 | "only certain GPUs with >= 24GB RAM are supported" (`infra/charts/sp1-cluster/values-example.yaml`); one task holds one whole card | yes: one GPU node process per card (`gpu{0..7}` services in the docker-compose deployment page); the local server itself supports device 0 only (`task.rs:160`, "only device 0 is supported at the moment"), one server per `CUDA_VISIBLE_DEVICES` | the split is by core shard, and the adopted shard is ONE core shard (section 1.3), so there is nothing to split across cards; the floor per card is unchanged. The cluster is a throughput tool, and its code is BSL |
| RISC Zero Bento (Boundless) | by segment onto Redis; `gpu_prove_agent` spawns one prove agent per card with `CUDA_VISIBLE_DEVICES`; the same `SEGMENT_SIZE` for every card, "the lowest common denominator"; joins form a tree (docs.boundless.network, performance-optimization and bento pages; `compose.yml:62-113`) | by `SEGMENT_SIZE` (2.1): 8 GB po2 19, 16 GB po2 20 | yes, documented: one 16 GB card 264 kHz, two 431 kHz (sub-linear, "bound by bus bandwidth, memory") | **the one documented configuration**: four 12 GB cards at po2 19 or 20 take segments off one queue and the joins fold them; the per-card floor is the segment, and the cost of small segments is the lift and join count (9 lifts and 8 joins for the adopted shard at po2 19) |
| RISC Zero `RISC0_PROVER=actor` | one process, several cards, a token budget per card from measured memory per po2 (`r0vm/src/actors/factory.rs`) | per po2 | yes, experimental since 3.0.1 | the same model without Bento's services |
| Pico Prism 2.0 | a global task queue across two machines, 16 x 5090, 100 Gbps between them (Brevis blog, May 2026) | not published | yes | no figure |
| OpenVM | metered execution on the CPU, segments to GPUs, an aggregation tree, "clusters with hundreds of GPUs" (docs, distributed-proving page) | 24 GB | yes | no 12 GB path |
| ZisK | coordinator and stateless workers; "splits the trace into pieces, proves each in parallel on separate machines, and aggregates"; the first worker aggregates a binary tree (docs, distributed execution page) | not documented | yes, `--gpu` per worker | no figure |
| Ceno | shards round-robin by `shard_id % device_count`; a shard never split across cards; one CUDA context per device (PR #1403) | not stated | yes | the same model |
| Column-split of one trace across cards (FRIttata eprint 2025/1285, HyperFond 2025/1349, deVirgo arXiv 2210.00264, Pianist 2023/1271, Cirrus 2024/1873, SumFold 2025/1653) | the sumcheck or FRI itself is distributed, each worker holding a slice of the columns or rows and exchanging small messages | a slice | research code or CPU clusters only | nothing shipped; the one route that would let four 12 GB cards hold what one 32 GB card holds for a SINGLE core shard, and nobody has it in a zkVM |
The reading. Every shipping system splits by rows (segments, shards, chunks), proves each on one card, and folds
with recursion. So "four 12 GB cards do what one 32 GB card does" is true for throughput (four shards in flight, or
four segments of one shard, then a join tree) and false for a single unit that exceeds one card: that unit must be cut
smaller, by the zkVM's segment knob (RISC Zero) or by our planner (`S_p`). For Igneum the units are already small and
independent (a shard, assigned by sortition), so the rig's natural mode is one prover process per card, each taking
its own shard. The distributed route therefore costs nothing in protocol and lands as an app change (section 3, route C).
### 2.6 Proof systems that run on AMD or Apple
| Target | What exists | Status | Source |
|---|---|---|---|
| Apple, RISC Zero Metal | the full STARK prover (rv32im, keccak, recursion) on Metal, automatic on Apple silicon; the Groth16 wrap x86 only | shipped, maintained (PR #3761's June 2026 matrix lists "metal (Mac M-series): build, run"); the only speed published is a 2023 M2 datasheet (14 to 93 kHz, approximate); nothing for M3, M4 or M5 | `risc0/sys/kernels/zkp/metal/`, dev.risczero.com local-proving page |
| Apple, ICICLE Metal (Ingonyama) | MSM, NTT, sumcheck on Metal since v3.6.0 (Mar 2025); "missing API implementations for Poseidon and Poseidon2 hashes, Merkle tree" at that release; v4.0.0 of 11 Jul 2025 is the latest | a library, closed-source backends under a free research licence (dev.ingonyama.com, install_gpu_backend page); no STARK prover built on it for Metal | ICICLE releases, the Metal blog |
| Apple, Jolt Metal | PR #1938 merged 30 Sep 2026 (the runtime and field kernels); PR #1733, the prover itself, a draft: M5 Max 2^25 cycles in 19.8 s, 3.2x over its CPU | the fastest Apple number anyone has published, in a draft | github.com/a16z/jolt pulls 1733 and 1938 |
| Apple, Stwo | CPU SIMD with NEON; ICICLE-Stwo promises Metal | CPU path shipped; no RISC-V guest of its own | stwo README, Ingonyama blog |
| Apple, Miden | `miden-gpu` on Metal | Cairo-class VM, not RISC-V | hackmd (bobbinth) |
| AMD, sppark | "A limited support for AMD's RDNA and CDNA GPUs" (README); SP1's tree carries no HIP build | a library | github.com/supranational/sppark |
| AMD, SP1 PR #2668 | an external port to RDNA3 and RDNA4 with "a caching memory allocator to work around hipMallocAsync leak bug" | **closed unmerged 20 Mar 2026** | github.com/succinctlabs/sp1/pull/2668 |
| AMD, OpenVM stark-backend HIP fork | `cuda2hip.hpp` so the same `.cu` builds under nvcc and hipcc, native `mont32_t.hip`, tested on gfx1100, targets MI300X and 7900 XTX | merged 15 Sep 2026 in a fork (Okm165/stark-backend PR #2), not upstream | the PR |
| AMD, Goldilocks NTT and STARK on ROCm | 19.19 ms NTT at 2^27 on an RX 7900 XTX; a Goldilocks STARK backend on HIP | research posts | ethresear.ch, qingming-g64-ntt and stark-g64 |
| Vulkan and WebGPU | ICICLE's Vulkan build (Jan 2025) with no installable backend; zkSecurity's WebGPU Stwo (5x on constraint evaluation, 2x end to end, no 64-bit integers in WGSL); ZPrize WebGPU MSM | prototypes; nothing proves a RISC-V shard | the pages named |
Said plainly, as `docs/analysis/amd-proving.md` said it: on 5 October 2026 no zkVM proves on an AMD GPU, and the only
Apple prover that ships is RISC Zero's. The AMD work that exists is two ports of CUDA STARK kernels through a HIP shim,
one closed, one in a fork; both are days of agent work to revive against a given tree, and PC 1's RX 9070 XT (gfx1201)
is the card to measure on.
## 3. For each route: the change to our guest, the aggregator and the node's verifier; the cost; the risk; 12 GB under 60 s
What the node verifies today: SP1 compressed proofs through `igneum-prove-host --mode verify` and `verify-segment`
(`vendor/igneum-node-pv1/igneum/exec/src/proving.rs:41-46, 878-926`), the pinned ids read at start and named in the
native statement (`program_ids`, `IGNEUM_PROOF_PROGRAM_IDS`), the record bound in a BLS-signed `ProofRecord` (version
1) or `SegmentRecord` (version 2) with the proof's SHA-256 (spec 7.7 item 1, 7.8 item 3). A different proof system
means a new `ProofSystem` version (design 5.6), a new pinned id, a second verifier command, and the record's version
field telling the node which. The swap procedure of design 5.6 (test vectors, 90% signalling, a 3-month overlap with
both verifiers, a wrap of the last old proof) is the path for any of the rows below that change the family.
The 60-s test. The litepaper's minute, the launch target of 20 to 60 s behind the tip, and the mine-and-prove
measurement that a shared card proves 3 to 4x slower (bench-log, `chain-pc2-pv1c`). No 12 GB card has run any
prover in this repository; the 12 GB times below are approximate, scaled from the 5090 by memory bandwidth (an RTX
3060 at 360 GB/s and an RTX 4070 at 504 GB/s against the 5090's 1,792 GB/s, NVIDIA's published figures, approximate),
which is the term a STARK prover is bound by. They are the numbers the first 3060-class run replaces.
| Route | Guest | Aggregator | Node verifier and record | Cost (agent time) | Risk | Reaches 12 GB with a real shard under 60 s? |
|---|---|---|---|---|---|---|
| **A. Re-size SP1's GPU server** (the prover-floor agent's patch, running tonight): remove the 20 GB panic (`builder.rs:37`), size `max_trace_size` to the shard (honour `ELEMENT_THRESHOLD`, or set the core allocation from the shard's measured cells), one core worker and a buffer of 1, a release threshold so the pool returns memory between stages, `drop_ldes` on; build with `CUDA_ARCHS` for Ampere, Ada and Blackwell | none: the same ELF, the same pinned id (the verifying key hashes the program and its preprocessed tables, not the server's buffer sizes; `HEIGHT_THRESHOLD` only shortens tables below the verifier's 2^22 maximum) | none: the compressed proof format and the aggregator guest are unchanged | none: the same `--mode verify`; the record format unchanged | hours to one day: a fork of `sp1-gpu/crates/prover_components` and `jagged_tracegen` (Apache or MIT), the 11-min cross-build, a per-card profile in `provedefault.rs`, a CI check that the fork's constants match the pinned verifier's | low on the protocol, medium on the build: the fixed recursion stage may hold the floor near 8 to 9 GB (section 1.3, approximate) and the first measurement says whether 11 GB is reached; a fork of `sp1-gpu` to carry forward on every SP1 release; the server rejects nothing it cannot hold, so an out-of-memory shard must fail cleanly and be left (the pool's rule today) | **memory: likely for the adopted shard** (10 to 11 GB on paper, section 1.4), **not** for the prototype shard (28 GB of live trace). **Time: prove-only yes** (4.3 s on the 5090 scales to about 15 to 22 s on a 3060 and 10 to 15 s on a 4070, approximate); **mine-and-prove on a 12 GB card: no at `S_p`** (3 to 4x on a shared card puts a 3060 at 45 to 90 s, approximate, and the miner's 1.7 GB on top of 11 GB does not fit), yes at `S_p/2` on a 4070 if the floor lands under 9 GB (approximate). The measurement decides; this is the route the gate waits on |
| **B. Halve `S_p`** (30,000 to 15,000 pgas, the fee-switch pattern): more and smaller shards | none | none: one deferred proof per shard, so 2x the shards per block; the chained aggregation stays one per block (9.7 s mining, 2.5 s alone) | none | hours: a fee-table change and a rollout plan like `fee-switch-devnet.md` | low: more records per block (the coinbase carries at most 8 shard records, spec 7.7 item 2, so `B_p / S_p` must stay at 8 or under); the assignment window and sortition unchanged | **alone, no**: the floor is the server's (13.9 GB at 0 cycles). **With A, it is the dial** that moves a 12 GB card from prove-only to mine-and-prove, and a 16 GB card to a comfortable fit |
| **C. One prover process per card on a rig** (the distributed route): the app runs one `sp1-gpu-server` per NVIDIA card (`CUDA_VISIBLE_DEVICES`, the per-device socket of `sp1-cuda/src/client.rs:211`, `.cuda().with_device_id(n)`), one host process per card, each taking its own assigned shard; the rig installer already picks cards (`igneum-rig-lib.sh`, `prover_decision`) | none | none: shards are independent units by design (spec 7.2); the aggregator runs on the biggest card | none | one day: the app's prover loop per card (`prover.rs` runs one loop today), the Settings and tile per card, the rig installer's prover unit per card, the socket cleanup per device (the root-socket rule of 5 October) | low; the throughput is per card, the host RAM 6 GB of pinned buffers per server (section 1.2), so a 4-card rig needs 32 GB of RAM or route A's smaller buffers | **it does not move the floor**: each card still needs A. It is the route that makes four 12 GB cards worth four shards a cycle, and it ships with A, not instead of it. Splitting ONE shard across cards is not a route: the adopted shard is one core shard (2.5), and column-split provers are research |
| **D. RISC Zero as proof system version 2** (CUDA and Metal; segments at po2 19 or 20) | a second guest: `core/` is plain Rust and ports as is; the precompile patches differ (SP1's `sha3` and `k256` patches against RISC Zero's `sha2`, `k256` and keccak circuit); the shard statement bytes unchanged; a second pinned ELF and image id in `elf/manifest.json` | a RISC Zero aggregator guest using composition (`env::verify` of the shard receipts, dev.risczero.com composition page); the chain rule (N verifies N-1) inside the family; **a block's shards must be one family**, and a chain cannot cross families inside the proof: a family switch lands at a segment boundary as a fresh chain (spec 7.8 item 6 already allows one after an unproven segment; the rule gains "or at a proof-system version change") | a second verifier mode (`--mode verify-r0`, the `risc0-zkvm` verifier, pure Rust, about 100 ms, 222 KB receipts); the record's `version` selects the family; the native statement names the family's pinned id; both verifiers in the node through the overlap of design 5.6 | 3 to 4 days: guest port and pinning 1, aggregator and chain rule 1, node verifier and record version 1, app profile and host modes 0.5, test vectors and the fast-time harness 0.5; plus the measurement day on PC 2 | medium: two proof systems in consensus for the overlap; RISC Zero's Groth16 wrap is x86 only (the light-client path of ledger P3 stays on SP1 or waits); a 222 KB receipt per shard against 1.27 MB today is a gain; the recursion tree per shard (9 lifts and 8 joins at po2 19) is extra time on small cards; `main` is at 5.0.0 with no release body, so the pin is 3.0.6 | **memory: yes by documentation** (po2 19 for an 8 GB card, po2 20 for 16 GB; 9 to 10 GB per 1 M cycles), the first documented sub-12 GB prover. **Time: approximate**: a 4090 does 808 kHz at po2 21, so the adopted shard is about 6 s on a 4090-class card and about 20 to 30 s on a 3060-class one at po2 19, prove-only; beside the miner over 60 s on a 3060, near it on a 4070. The prover-floor agent's PC 2 run is the first real number |
| **E. Airbender, OpenVM, ZisK, Pico, Ziren** as version 2 | a new guest each (RISC-V, except Ziren's MIPS); OpenVM's and ZisK's toolchains are the most complete | each has its own recursion; OpenVM's aggregation and Halo2 wrap are the most documented | a new verifier each (STARK under 300 KB for OpenVM; PLONK or FFLONK for ZisK and Airbender) | 4 to 6 days each | the same two-family cost as D with no memory gain: 21 GiB (Airbender), 24 GB (OpenVM, Ziren), undocumented (ZisK, Pico); Pico's and Ziren's GPU code is BUSL or closed | **no**: none documents a floor under 21 GiB; the race is tuned for 5090 clusters |
| **F. Jolt (Lattice Jolt) as version 2**: a sumcheck prover with no codeword; CPU and Metal | a new guest (RV64IMAC, Jolt's toolchain; no keccak precompile today, approximate, so the trie hashing costs more cycles than in SP1) | **none exists**: no recursion or continuation shipped, so the aggregator would verify N shard proofs natively and the chain rule would live in the native statement until Jolt's recursion lands | a Dory verifier (BN254 pairings, about 50 KB, sub-second, approximate) or an Akita verifier (lattice, 65 to 80 KB); no on-chain verifier shipped | 5 to 8 days for the guest, the verifier and the record; the aggregator question has no answer in the code | high: alpha software, no audit, no production user, no recursion; the proof system of the miner's CPU, not of its card | **memory: yes by a wide margin** (about 0.9 GB for the adopted shard at 200 bytes a cycle, approximate). **Time on a CPU: about 2 to 3 s** for 4.7 M cycles at over 2 M cycles a second (a16z, Sep 2026, laptop CPU; approximate for our guest), **on Metal under 1 s** (PR #1733's 2^25 in 19.8 s on an M5 Max, approximate). The numbers are the best in this document and the software is not shippable |
| **G. Folding** (Nova family, lattice folding) | a step circuit per transaction or per opcode group | a decider per shard | pairing or lattice verifier | weeks of research, no code over our field | the family that Nexus left | **no** today; the watch item for 2027 |
| **H. AMD through a HIP port of SP1's kernels** (PR #2668 revived against 6.8.1, or the `cuda2hip` shim of the OpenVM fork) | none | none | none: the same SP1 proofs | 3 to 5 days plus PC 1's RX 9070 XT to measure; the `hipMallocAsync` leak needs the caching allocator the PR carried | medium: a kernel port with no upstream; the sppark NTT has a limited HIP path and cuPQC none | memory as route A (the same buffers); **time unmeasured on any AMD card**; the one route that gives AMD miners the 20% pool share |
| **I. Apple through RISC Zero Metal** (route D's Metal half) | as D | as D | as D | inside D's 3 to 4 days | the 2023 M2 figure (14 kHz, approximate) says 5 minutes for the adopted shard; an M5 Max is not measured by anyone | **memory: yes** (unified memory, 64 GB on the M5 Max). **Time: unknown**; the Mac measure lock run is the number |
## 4. The ranked recommendation
| Rank | Route | Why | Gate |
|---|---|---|---|
| **1. Soonest to 12 GB with the least change: A, with B as the dial and C for rigs** | re-size SP1's GPU server; keep the guest, the aggregator, the verifier and the pinned ids exactly as they are; set `S_p` from the first 12 GB measurement; one prover per card on rigs | nothing in consensus moves; the work is a fork of two Apache crates and an app profile; it is already running tonight; every other route costs days and adds a second verifier | the prover-floor agent's rows: the adopted shard under 11 GB alone and the time on the first 3060-class or 4070-class card, prove-only and beside the miner. If under 11 GB and under 60 s prove-only: ship 0.3.12 with the 12 GB tier as prove-only and `S_p/2` measured for mine-and-prove. If not under 11 GB: route D |
| **2. The fallback if A misses 11 GB, and the Apple route either way: D, RISC Zero as version 2** | the only shipped prover with a documented sub-12 GB configuration and a shipped Metal path; Apache or MIT including the kernels; 222 KB receipts | the swappable interface was built for this (design 5.6) and the node already names the pinned id in the statement, so a second family is a version, not a redesign; the cost is 3 to 4 days plus the overlap | PC 2's po2 19 and 20 rows (memory, time per segment, lift and join) tonight; the Mac's Metal row |
| **3. Best in five years: the sumcheck family without a codeword (Jolt-class), or the sumcheck-plus-WHIR family SP1 and OpenVM already converge on** | Jolt proves the adopted shard in seconds on a laptop CPU at under 1 GB of memory, which is the only route that gives AMD-only, Apple and 8 GB machines the proving share with their existing hardware; its verifier is small (50 to 80 KB); its licence is MIT or Apache. It is alpha with no recursion, so not before it ships a stable release with continuations and an audit. SP1 Hypercube and OpenVM SWIRL are the same mathematics with a hash-based PCS and a GPU today, which is why staying on SP1 now loses nothing in that direction | do not adopt now; re-read Jolt and the Arc or WARP accumulation line at every 6-month era draw (design 5.6's swap procedure needs 90% signalling and a 3-month overlap, so the lead time is the schedule) | a stable Jolt tag with recursion, an audit, and a CUDA or merged Metal prover |
| **The interface question** | yes: `ProofSystem` is versioned (`VERSION`, `program_id`, `verify_segment`), the record carries `version`, the node reads pinned ids at start and names them in the native statement, and the overlap procedure keeps both verifiers in the node for 3 months with `B_p` from the stricter table. What is missing for two families at once is small and named: the record version selecting the verifier command, the fresh-chain rule at a version change, and the shard plan carrying the family per block so a block's shards are homogeneous (the aggregator folds one family). Those three items are in route D's day of node work | so the answer to "ship one now and move to the other later" is yes, and route A ships nothing that has to be undone | |
The honest statement of what this ranking does not know: no 12 GB card has run any prover here. Route A's time
figures are bandwidth scaling, labelled approximate; route D's are a 4090 figure scaled the same way. The first 3060
or 4070 in this repository replaces both columns, and the plan is to borrow or buy one this week (a 4070 is the
common 12 GB card of 2026; a 3060 the common older one; both are the gate's named class, design R2).
## 5. The tier consequences, and the public line while the change is made
Every number carries its consequences (CLAUDE.md, 5 October 2026). The table says what each tier has today on SP1
6.8.1, what route 1 (A plus B plus C) gives it if the gate is met, what route 2 (D) adds, and what only route 3 would
give. "Today" is measured; the rest is the routes' expected outcome, labelled, until the measurement.
| Tier | Today (measured, bench-log 5 October) | Route 1: re-sized SP1 server, `S_p` as the dial, one server per card | Route 2: RISC Zero version 2 | Only route 3 (sumcheck without a codeword) |
|---|---|---|---|---|
| Home miner, one 8 GB NVIDIA card | mines; proves nothing (the server panics under 20 GB) | proves nothing at `S_p` (the floor's fixed terms, 8 to 9 GB approximate, leave no room); perhaps empty shards | prove-only at po2 19 (Boundless' 8 GB tier), the miner paused per shard; time approximate 30 to 60 s | mines and proves on its CPU |
| Home miner, one 12 GB card (3060, 4070) | mines; proves nothing; the litepaper's gate card | **prove-only at `S_p`** if the floor lands under 11 GB (expected, section 1.4): about 15 to 22 s a shard, approximate; **mine-and-prove at `S_p/2`** on a 4070 if the floor is under 9 GB, approximate; on a 3060 the shared card misses 60 s, approximate, so its default is prove-only with the miner paused per shard (the 16 GB rule of `provedefault.rs` today, moved down a tier) | prove-only at po2 20 (16 GB tier) or po2 19; mine-and-prove not inside 60 s on a 3060, approximate | mines and proves, CPU |
| Home miner, one 16 GB card (5080, 4080, 4060 Ti 16 GB) | an empty shard alone (13.9 GB); nothing beside the miner | **mine-and-prove at `S_p`** (11 GB plus the miner's 1.7 GB), about 7 to 12 s a shard alone and 20 to 40 s beside the miner, approximate | mine-and-prove at po2 20 | the same |
| Home miner, one 24 GB card (4090, 3090) | the adopted shard alone (20.4 GB) and beside the miner (22.2 GB, approximate for the card); the prototype shard never | mine-and-prove at `S_p` with 10 GB to spare; the prototype shard (28 GB live) only if the devnet's fee switch has passed, which it has from DAA 210,000 | the same with Metal irrelevant | the same |
| Home miner, one 32 GB card (5090) | everything, measured | everything, with more shards in flight if the pool releases between stages | the same | the same |
| Rig, several NVIDIA cards | one prover on the biggest card (`prover_decision`) | **one server per card**, each its own shard; the aggregator on the biggest card; host RAM 6 GB pinned per server today, under 2 GB with route A's buffers | the same model (Bento's) | the same |
| Pool user | through the pool; who proves is open (spec 09) | unchanged | unchanged | unchanged |
| AMD-only (RX 9070 XT, 7900 XTX) | mines; proves nothing on the card; the CPU path 282 s a shard at 30 GB | unchanged until route H (a HIP port, 3 to 5 days, measured on PC 1's 9070 XT) | unchanged: RISC Zero is CUDA and Metal only | mines and proves on its CPU |
| Apple silicon (M-series) | mines (26.7 MH/s on the M5 Max); the SP1 CPU prover 41 to 55 s for an empty shard, 272 s for a small one | unchanged | **proves on the GPU through Metal** (64 GB unified memory on an M5 Max holds any segment); the time is the measurement | proves in seconds on Metal (Jolt's draft PR figure, approximate) |
| Windows under 32 GB of RAM | off (the WSL2 prover held 7.9 GB) | the pinned buffers fall with `max_trace_size`, so a 16 GB PC likely qualifies, approximate; measure | RISC Zero's CUDA path also runs in WSL2 | |
The deadlines these fit (spec 7.2 item 3, the litepaper, `proving-v1.md`): the 10-s exclusive window is the 5090's
alone; a 12 GB card at 15 to 22 s proves its assigned shards in the open phase and is paid when no faster card took
them, which on a chain with few 5090s is most of the time; the minute of the litepaper holds for prove-only 12 GB
cards and for mine-and-prove 16 GB cards; the 600-s unproven deadline holds for every tier above the CPU path.
### The public line while the change is made
The litepaper's sentence today ("Target: shard size will be set so a 12 GB card proves one shard in about 20
seconds") is a target and says so (fud-ledger P1, overclaim 27). What this document adds, for `site/litepaper.html`,
`site/miner.html` and the app's Proving tile, in the copy law:
> Proving runs on NVIDIA cards with 24 GB or more today. A build for 12 GB and 16 GB cards is being measured: the
> memory is the prover's buffers, not the shard, and the fix is a smaller build of the same prover. AMD and Apple
> cards mine. A second prover with an Apple path exists and is the fallback.
And the rule for the next status line, whichever way the measurement goes: the number, the card it was taken on, and
the tier it moves, in one sentence, the day it is taken.
### What this document does about it
| Consequence | Action | Owner |
|---|---|---|
| The gate card has never run a prover here | get a 4070 or 3060 into the measurement loop this week; until then every 12 GB figure stays approximate | coordinator; Josh for the card |
| Route A's gate | the prover-floor agent's rows (asked for by message tonight); if under 11 GB, `provedefault.rs` gains the 12 GB prove-only and 16 GB mine-and-prove tiers and the rig installer one server per card | prover-floor agent, then the proving engineer |
| Route D's measurement | RISC Zero 3.0.6 at po2 19 and 20 on PC 2 (CUDA) and on this Mac (Metal), the same shard statement run natively: memory, time per segment, lift and join, receipt size | prover-floor agent (PC 2); a Mac measure job for Metal |
| The two-family node items (record version selects the verifier, fresh chain at a version change, one family per block) | spec 7.8 gains the three rules when route D starts; nothing changes before | execution engineer |
| AMD | route H is a 3-to-5-day job with a measurement on PC 1's 9070 XT; opened as a plan when route A's result is in | execution engineer |
| The public line | the paragraph above to the site and the tile with the next site pass | site-pages owner |
## Sources
Our own: `docs/bench-log.md` entries "proving v1: segment records, the chain rule, the unproven rule" (5 October 2026),
"the SP1 CPU prover on PC 1" (5 October), "shard proving on the RTX 5090" (4 October); `docs/plans/proving-v0.md`,
`proving-v1.md`; `docs/analysis/amd-proving.md`; `docs/spec/07-execution.md` 7.2, 7.6, 7.7, 7.8; `docs/design/execution-layer.md`
5.1 to 5.7; `proving/igneum-prove` (`host/src/proof_system.rs`, `program/src/main.rs`, `aggregator/src/main.rs`,
`elf/manifest.json`); `vendor/igneum-node-pv1/igneum/exec/src/proving.rs`; `app/igneum-app/src/provedefault.rs`, `prover.rs`.
SP1 6.8.1, read from `~/.cargo/registry/src/index.crates.io-*/` and the vendored tree `vendor/sp1-6.8.1` (commit
c84ada1e, 24 Sep 2026) on the `prover-floor` worktree: `sp1-core-executor-6.8.1/src/opts.rs`, `src/utils.rs`,
`src/artifacts/rv64im_costs.json`; `sp1-prover-6.8.1/src/components.rs`, `src/worker/config.rs`, `src/shapes.rs`;
`sp1-primitives-6.8.1/src/fri_params.rs`; `sp1-verifier-6.8.1/src/compressed/config.rs`; `sp1-hypercube-6.8.1/src/verifier/config.rs`;
`sp1-cuda-6.8.1/src/server.rs`, `src/client.rs`; `sp1-gpu/README.md`, `sp1-gpu/crates/prover_components/src/builder.rs`,
`src/components.rs`, `sp1-gpu/crates/jagged_tracegen/src/lib.rs`, `sp1-gpu/crates/shard_prover/src/prover.rs`,
`sp1-gpu/crates/cuda/src/task.rs`, `src/device.rs`, `sp1-gpu/crates/sys/lib/runtime/mem_pool.cu`, `sp1-gpu/crates/zerocheck/src/primitives.rs`.
Web: docs.succinct.xyz (hardware-acceleration, hardware-requirements, proof-types, security-model, provers introduction,
cluster architecture, docker-compose deployment); blog.succinct.xyz (sp1-hypercube, real-time-proving-16-gpus,
sp1-hypercube-is-now-live-on-mainnet); github.com/succinctlabs/sp1 releases v6.0.0 to v6.8.1, issues #2674, #2930,
#2950, #2969, pulls #2631, #2668, #2723, #2917, #2974; github.com/succinctlabs/sp1-cluster (README, LICENSE,
`infra/charts/sp1-cluster/values-example.yaml`, `crates/worker/src/config.rs`); eprint 2025/917 (jagged polynomial commitments).
RISC Zero: `~/.cargo/registry` crates `risc0-zkp-3.0.4/src/lib.rs`, `risc0-zkvm-3.0.4/src/receipt.rs`, `src/host/recursion/prove/mod.rs`,
`risc0-circuit-rv32im-4.0.4/src/execute/mod.rs`, `src/zirgen/defs.rs.inc`, `src/prove/hal/cuda.rs`; github.com/risc0/risc0
`risc0/zkvm/src/host/client/env.rs`, `risc0/zkvm/Cargo.toml`, `risc0/zkvm/build.rs`, `risc0/sys/kernels/zkp/{cuda,metal}/`,
`risc0/r0vm/src/actors/factory.rs`, `risc0/circuit/recursion/src/lib.rs`, releases v2.0.0, v3.0.1, v3.0.6, pull #3761;
dev.risczero.com (local-proving, composition); docs.boundless.network (bento, performance-optimization, quick-start);
github.com/boundless-xyz/boundless (`compose.yml`, `bento/README.md`, `bento/LICENSE-BSL`); github.com/ekrembal/gsr-stark-verifier pull 5; l2beat.com/zk-catalog/risc0.
Others: zksync.io/airbender, docs.zksync.io airbender and proving pages, github.com/matter-labs/zksync-airbender (README,
`docs/gpu.md`, pull #448), veridise.com (the Airbender audit); 0xpolygonhermez.github.io/zisk (introduction, limits,
distributed execution, installation), github.com/0xPolygonHermez/zisk (README, pull #1238, `zisk-contracts`);
blog.openvm.dev (2.0, 2.0-production, 2.1, openvm-gpu, v1), docs.openvm.dev (security-model, distributed-proving, sdk);
pico-docs.brevis.network, github.com/brevis-network/pico and pico-gpu (README, LICENSE), blog.brevis.network (Prism 1.0,
2.0, 2.1); docs.zkm.io (prover, performance), github.com/ProjectZKM/Ziren, zkm.io (the independent evaluation of v1.1.4),
eprint 2026/2330; github.com/starkware-libs/stwo and stwo-cairo (README), ingonyama.com (ICICLE-Stwo, the Starknet
partnership, ICICLE Metal v3.6), dev.ingonyama.com (install_gpu_backend), blog.zksecurity.xyz/posts/webgpu, starkware.co
(S-two 2.0.0, Nexus on S-two), theblock.co (S-two on Starknet); github.com/a16z/jolt (README, book: intro, dory, akita,
streaming, recursion, blindfold; pulls #1733, #1938; tags), a16zcrypto.substack.com ("How to prove software ran
correctly", Sep 2026), a16zcrypto.com (jolt-6x-speedup, 64-bit-proving-jolt, zkvm-jolt-zero-knowledge, faqs-on-jolts-initial-implementation),
eprint 2025/611; github.com/scroll-tech/ceno (README, Cargo.toml, pull #1403), ceno-gpu-mock, scroll.io (Ceno post),
osec.io ("zkVMs' unfaithful claims"); github.com/nexus-xyz/nexus-zkvm (README, LICENSE), blog.nexus.xyz (roadmap);
lita.gitbook.io (Valida architecture, benchmarks); github.com/powdr-labs/powdr; irreducible.com (announcing-binius64,
reinventing-irreducible, irreducible-shutting-down), github.com/binius-zk/binius64, eprint 2026/1656; eprint 2021/1043,
2022/1010, 2024/1609, 2025/1187, 2024/1586, 2024/185 (linear-code commitments, WHIR, Vortex), github.com/Consensys/linea-monorepo;
PolyhedraZK/Expander and blog.polyhedra.network (returned 530 tonight); eprint 2021/370, 2024/2099, 2024/1220, 2024/416,
2024/1605, 2025/247, 2025/294, 2026/242, 2024/1731, 2025/753, 2026/1371 (folding and accumulation), sonobe.pse.dev,
github.com/privacy-scaling-explorations/sonobe, NethermindEth/latticefold, LFDT-Nightstream/Nightstream; eprint 2023/1271,
2024/1208, 2024/1873, 2025/1349, 2025/1653, 2025/1285, 2018/691, arXiv 2210.00264, 2602.16338 (distributed proving);
github.com/supranational/sppark, github.com/Okm165/stark-backend pull 2, ethresear.ch (qingming G64 NTT and STARK on ROCm);
github.com/cysic-labs/venus, erigon.tech (Zilkworm), github.com/DelphinusLab/prover-node-docker, hackmd.io/@bobbinth
(Miden), ethproofs.org/clusters (5 October 2026).

View file

@ -0,0 +1,408 @@
# Layer 3 soundness: the per-warp scratch with read-modify-writes
5 October 2026 (night), cryptographer role, Counter ASIC 2.0 plan step 4 (`docs/plans/counter-asic-2.md`). Branch
`ca2-soundness` on top of `readwidth` b970dda (the scratch as a class parameter, 32 or 128 KiB per warp). Tests:
`igneum-pow/tests/scratch.rs`; Metal runs through `proto-metal/packbench` on the M5 Max; commands and counts in
`docs/bench-log.md` (entry of the same date). Nothing here touches the lottery hash as shipped: variant 5 is behind
`LoadClass::scratch(k, kb)` and is never emitted by generator version 2.
Every figure below is measured (machine, date, command named) or cited; "approximate" marks a figure from memory.
## 0. The five findings
| # | Question | Finding | Status |
|---|---|---|---|
| 1 | Is what is written uniform and beyond a chip's precomputation? | The fill is a bijection of the lane nonce, the rewrite a bijection of the fold value in each word; written words show no bit bias over 3 to 12 million rewrites per class (worst 3.63 sigma of 6). The fill IS precomputable, by design, and at 64 slots 78.5 percent of reads are fill reads. | sound as a function; see 2 for what that means |
| 2 | Does any short cut avoid the writes? | No short cut inside a unit: a slot after d read-modify-writes needs all d fold values (replay test). But the live state is bounded by the read-modify-write count, not by the scratch size, because CPU verification resets the scratch per unit: 64 to 320 bytes per lane at scr2 to scr8, whatever the nominal 32 KiB, 128 KiB or 1 MiB. The named chip (cache mirror plus recompute) keeps that in SRAM at under 5 percent of its mirror and its gain does not move at any share under the 6 GB cap. | NOT sound as an anti-chip layer |
| 3 | Is the verifier's one-warp simulation exact? | Exact when the GPU's lazy per-unit tag is unique over the arena's life and the arena holds no stale tag. The kernels rely on this and neither host guarantees it (no clear at allocation, no clear at the 32-bit wrap of the tag counter, 16.4 minutes on a 5090). With the host contract of section 4.3 the simulation is exact: 14 edge packs twice, 200 fuzz packs, consecutive units on one warp and the wrap inside a launch all match the CPU on Metal (228 of 228); a broken tag and a broken fill are caught (3 of 3). | sound with a host contract; today it is luck |
| 4 | The attack surface of the writes | Out of bounds: impossible by the mask, 42 of 42 emitted kernels pass the static check, which catches six deliberate breaks. Aliasing: none, lane-major arenas disjoint by (warp, lane), two logical units of a wave64 get two arenas. Ordering: one lane, one slot, program order; no cross-lane sharing, no atomics needed. Alignment: 16-byte slots at 16-byte offsets from a 256-byte-aligned base. Wrap: identical to the CPU, tested at the launch level. | sound |
| 5 | What a conformance vector must carry | The class and geometry, the fill and rewrite, the host contract (tags, clearing, groups a multiple of warps), two consecutive units on one warp with a forced slot collision, a unit in the top 256 nonces with the wrap inside the launch, and the fingerprint declared independent of the warp count. The standard three-unit vectors catch a broken tag only through base 1,000,000 and would miss it at a 1 MiB scratch. | defined in section 6 |
Recommendation (section 10): do not adopt layer 3 as the plan states it (read-modify-writes taken from the 16
dataset loads). It replaces latency-bound dataset reads with cache-bound ones for the GPU, costs the named chip
nothing it cannot keep in a few megabytes of SRAM, and leaves that chip's gain at 2.4x at every share. The lever
that moves that chip is the mixer multiplier of the M16 analysis (x2 brings it to 1.2x, x4 to 0.6x, under the
verifier's 10 ms gate). If a scratch is kept for another reason, add the read-modify-writes beside the 128 loads,
never in their place, and ship the host contract and the vector of section 6 with it.
## 1. What the branch implements
| Piece | Where | What |
|---|---|---|
| Class | `igneum-pow/src/generator.rs:170-230` | `LoadClass { scratch: Some(k), scratch_kb }`: `k` of the 16 memory slots are `Op::Scratch`; `scratch_kb` KiB per warp of 16-byte slots, lane-major, `slots = kb x 2` per lane (32 KiB: 64, 128 KiB: 256); `scratch_slot_mask() = slots - 1` |
| Draw | `generator.rs:488-491` | the first `k` of the 16 drawn load slots become scratch ops (a uniform k-subset); the source register follows the fresh-source rule like a load |
| Fill | `igneum-pow/src/verify.rs:30` | `scratch_fill(seed, base, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot x 0x9e3779b1 + (j + 1) x 0x85ebca77)`, j in 0..2 |
| Fold | `verify.rs:18` | `x = dst ^ w0; x = (rotl(x, 11) x 0x9e3779b1) ^ w1; x = (rotl(x, 11) x 0x9e3779b1) ^ w2; dst = x` (the read-width fold over the three data words) |
| Rewrite | `verify.rs:41` | the slot becomes `(x ^ w1, rotl(x, 7) ^ w2, x + w0)` |
| CPU model | `verify.rs:48-100`, `:305-312` | `ScratchModel`: per (lane, slot) a written bit and three words; an unwritten slot reads as its fill; one model per unit, so a unit starts from the fill |
| Acceptance | `igneum-pow/src/accept.rs:202-215, 374` | a scratch site that reads one slot in all 32 lanes rejects the program (lane-constant site); scratch slots carry bit 31 in the address list and are left out of the distinct-address bound, which now covers the dataset loads only |
| GPU statement | `igneum-pow/src/emit.rs:143-157` | `s_ = rN & mask; v_ = 16-byte load of slot s_; m_ = (v_.x == tag) ? ~0 : 0; w = (v_.yzw & m_) \| (fill & ~m_); fold; dst = x_; 16-byte store of (tag, x_ ^ w1_, rotl(x_, 7) ^ w2_, x_ + w0_)` in Metal, CUDA and OpenCL |
| Persistent prologue | `emit.rs:159-175` | `lane = tid & 31; warp_ = tid >> 5; arena = scratch + (warp_ x 32 + lane) x words_per_lane; for (g_ = warp_; g_ < groups; g_ += nwarps_) { gbase = baseNonce + g_ x 32; tag = salt + g_; ... }` |
| Hosts | `proto-metal/packbench.swift:144-164`, `proto-opencl/host.c:1025-1033, 1268` | the arena is allocated and never written by the host; `salt` starts at 1 and advances by the launch's unit count; no clear at allocation, none at the wrap |
The constraint of the night (coordinator, 5 October 2026): the whole working set on an 8 GB card stays under 6 GB
(1 GiB table, the layer 5 hot table, the scratch of every resident warp, buffers), which caps the scratch at tens of
KiB per warp. On an RTX 5090 at full occupancy (170 SMs x 64 warps = 10,880 warps, approximate hardware maximum;
the measured version 2 kernel ran 24 warps per SM, 4,080 warps, `docs/bench-log.md` M11, 4 October 2026):
| Scratch per warp | 10,880 warps | 4,080 warps (measured occupancy) | Table + scratch at 10,880 | Under 6 GB with a 1 GiB table |
|---|---|---|---|---|
| 32 KiB | 340 MiB | 128 MiB | 1,364 MiB | yes |
| 128 KiB | 1,360 MiB | 510 MiB | 2,384 MiB | yes |
| 1 MiB (the first experiment) | 10,880 MiB | 4,080 MiB | 11,904 MiB | no |
## 2. Question 1: uniformity of what is written
### 2.1 As functions
The fill of word j of slot s for lane nonce n is `splitmix32(((n ^ seed[j]) + s x 0x9e3779b1 + (j + 1) x 0x85ebca77))`.
`splitmix32` is a bijection of its 32-bit input; for fixed (seed, s, j) the input is a bijection of n. So over any
2^32 consecutive nonces every 32-bit value appears once as the fill of (s, j): uniform. Test
`fill_is_a_bijection_of_the_nonce`: 2^16 consecutive nonces give 2^16 distinct words for 7 slots x 3 word
positions; the fill of lane l at base b equals the fill of lane 0 at base b + l; it wraps with the nonce
(base 0xffffffe0, lane 32 equals nonce 0).
The rewrite `(x ^ w1, rotl(x, 7) ^ w2, x + w0)` is, for fixed old content w, a bijection of the fold value x in
EACH word. Test `rewrite_is_a_bijection_of_the_fold_value`: 2^16 consecutive x give 2^16 distinct words in each
position for 16 random w. Consequence: a uniform x gives a uniform word in every position, and the three words
are three images of the same x, so a rewritten slot carries exactly 32 bits of new state behind 96 bits of
storage (from w and any one written word, x is recovered; the test checks all three inversions).
The fold value x is `fold(dst, w)`, a bijection of `dst` for fixed w (xor, then rotate-multiply-xor twice; the
multiplier is odd). So the written words are uniform whenever `dst` is, and `dst` is a register of the running
program.
### 2.2 The attack: what a chip can precompute
The fill is a pure function of (seed, nonce, slot): precomputable, and meant to be (the verifier computes it too).
A chip never stores a fill; it computes it in about 10 integer operations when a slot is first touched. The written
words depend on `dst`, the register state at that instruction, which depends on every earlier instruction of the
hash, including the dataset loads. Nothing about them is precomputable before the hash runs. This is the whole of
what question 1 can give: the writes are as unpredictable as the registers. What that is worth is question 2.
### 2.3 The stats run (the `TESTS.md` section 3 shape)
Test `written_words_unbiased_and_rehit_rates`, M5 Max, 5 October 2026, `cargo test --test scratch`: for each
class, programs of `igneum-genesis`, `igneum-genesis/stats1`, `igneum-genesis/stats2`, 2^11 units each (196,608
hashes per class), closed-form dataset, every read-modify-write traced (`verify::interpret_warp_scratch`). Ones
count per bit of every written word and of the change each rewrite makes (written XOR read), sigma = sqrt(N)/2,
limit 6 sigma like the acceptance rule's output check.
| Class | Slots per lane | RMW per hash per lane | Rewrites traced | Max bias, written words (sigma) | Max bias, written XOR read (sigma) |
|---|---|---|---|---|---|
| scr2k32 | 64 | 16 | 3,145,728 | 2.61 | 3.40 |
| scr4k32 | 64 | 32 | 6,291,456 | 2.18 | 3.81 |
| scr8k32 | 64 | 64 | 12,582,912 | 3.63 | 2.25 |
| scr2k128 | 256 | 16 | 3,145,728 | 3.36 | 2.19 |
| scr4k128 | 256 | 32 | 6,291,456 | 2.71 | 3.68 |
| scr8k128 | 256 | 64 | 12,582,912 | 2.73 | 2.60 |
576 bit positions (6 classes x 3 words x 32 bits) at under 4 sigma is what fair coins give. Verdict: no structural
bias in what is written. Like `TESTS.md` section 3 this is a sanity check, not a proof of strength.
## 3. Question 2: no short cut avoids the writes
### 3.1 Inside a unit: the chain is dependent
Slot s of lane l, touched d times in a unit, holds `w_d = rewrite(x_d, w_{d-1})`, `w_0 = fill`, with
`x_i = fold(dst_i, w_{i-1})`. `x_i` depends on the slot content before it, which depends on every earlier fold
value of that slot; and `dst_i` is the register state, which the earlier fold values entered. Test
`slot_is_replayable_from_its_fold_values`: a slot after 64 read-modify-writes is reproduced from the fill and the
64 fold values; dropping one diverges. So a chip cannot skip a write and still read the slot later. It has three
ways to hold a slot, all exact:
| Store | Bytes per lane | Cost on a re-hit |
|---|---|---|
| Dense: every slot, 12 data bytes plus a valid bit | 12 x slots: 776 (64 slots), 3,104 (256), 24,832 (2,048) | one SRAM read |
| Sparse: only touched slots, 12 bytes plus a slot index | about 13 x distinct: 185 to 820 (table below) | one lookup |
| Implicit: only the fold values, 4 bytes plus a slot index per read-modify-write, replay on a re-hit | 5 x 8k: 80 (scr2), 160 (scr4), 320 (scr8) | d rewrites of 5 integer ops |
The implicit store is smaller than the dense one whenever `slots > 8k / 3`: at scr4 above 10.7 slots, at scr8
above 21.3. So "the smallest scratch at which keeping it implicitly is dearer than storing it" is 8k/3 slots per
lane, 2.7 to 5.3 KiB per warp at scr4 to scr8. Every size on the table, 32 KiB and above, is past it: a chip
keeps the scratch implicitly in 80 to 320 bytes per lane at any nominal size, and the replay cost is bounded by
the re-hit depth, which the next table measures.
### 3.2 The re-hit rate at 64 and 256 slots (and at 2,048)
Measured in the same test run (every read-modify-write of 196,608 hashes per class traced; a re-hit is a read of a
slot the same unit wrote earlier). Birthday: `distinct = S (1 - (1 - 1/S)^n)` for n uniform draws from S slots.
| Class | S | n = RMW per hash | Distinct slots, birthday | Re-hits, birthday | Re-hit %, birthday | Re-hit %, measured | Max chain depth seen | Slot histogram against uniform |
|---|---|---|---|---|---|---|---|---|
| scr2k32 | 64 | 16 | 14.26 | 1.74 | 10.9 | 12.58 | 7 | chi2 z 22,023; hottest slot 2.74x, coldest 0.83x |
| scr4k32 | 64 | 32 | 25.33 | 6.67 | 20.8 | 21.47 | 8 | z 10,880; 1.87x, 0.92x |
| scr8k32 | 64 | 64 | 40.64 | 23.36 | 36.5 | 36.99 | 9 | z 7,587; 1.39x, 0.91x |
| scr2k128 | 256 | 16 | 15.54 | 0.46 | 2.9 | 3.84 | 5 | z 19,146; 5.10x, 0.82x |
| scr4k128 | 256 | 32 | 30.14 | 1.86 | 5.8 | 6.25 | 6 | z 9,632; 3.06x, 0.90x |
| scr8k128 | 256 | 64 | 56.72 | 7.28 | 11.4 | 11.89 | 6 | z 5,873; 1.98x, 0.90x |
| 1 MiB (not run) | 2,048 | 32 | 31.76 | 0.24 | 0.8 | | | |
Two readings. First, the slot a read-modify-write addresses is the low 6 or 8 bits of a program register, and
those bits are not uniform: `or` sets them, `mul` clears them, so one slot of 256 is addressed 5.1 times as often
as the mean and the re-hit rate runs 2 to 33 percent above the birthday rate. For the dataset the same bias on the
low bits of a 28-bit address is harmless (it moves the read inside an item); for a 64-slot scratch it concentrates
the chain. Second, the chain depth is small: at scr4k32 the deepest slot in 196,608 hashes saw 8 earlier
read-modify-writes; a replay costs at most 8 x 5 integer operations, against about 1,170 for one dataset item.
### 3.3 The live state is bounded by the read-modify-write count, not by the size
The verifier evaluates one unit from nothing but (program, day, nonce group): `ScratchModel::new` per unit,
`verify.rs:296`. Every conforming GPU must therefore start every unit from the fill, which the tag does
(section 4). So no state crosses a unit boundary, and the state a unit can ever read back is what it wrote itself:
at most 8k slots per lane. The nominal size only sets how often those 8k writes land on the same slot (the table
above). The scratch's "memory" is 8k x 16 bytes per lane of touched slots, 256 bytes to 1 KiB at scr2 to scr8,
and a chip holds it implicitly in 80 to 320 bytes.
The attack of rolling back or sharing scratch between units has nothing to take: a unit starts from the fill
whatever ran before it, so a chip that clears 64 valid bits per unit has rolled back, and nothing one unit wrote
is readable by another. The CPU verifier is that chip.
### 3.4 The named chip, and what the scratch costs it
The strongest chip the plan has priced (coordinator, 5 October 2026): the whole 256 MiB cache on the die, computing
every dataset item on the fly. Its cache SRAM, from `docs/analysis/sram-mirror.md` revision 2 (`ca2-analysis`
e6085c6), headline at shipped-product density / bit-cell lower bound, dollars per good die approximate: 164 / 83
mm^2 and $30 / $13 at N7 (shipped density from AMD 3D V-Cache, 64 MB on 41 mm^2, Hot Chips 2021); 128 / 64 mm^2 and
$46 / $21 at N5, N3E and Intel 18A (TSMC N5 HD macro 31.8 Mib/mm^2 after assist overhead, SemiAnalysis, December
2022); 106 / 54 mm^2 and $56 / $26 at N2; with a 96 MB hot table 226 / 114 at N7, 175 / 89 at N5, 146 / 74 at N2.
The chip's cache cost in the table below is the N5 headline, 128 mm^2 and $46 per good die. It computes every item
through the mixer (`docs/analysis/m16-recompute-attacker-2026-10-05.md`: 128 items per hash, about 1,170 integer
operations per item, 150,000 per hash; at a 50 T op/s integer budget equal to a 5090's, approximate, 0.33 Ghash/s).
Against the measured version 2 rate of the RTX 5090, 139.7 MH/s (`docs/bench-log.md` M11, 4 October 2026), that is
2.4x before any fixed-function factor, 7x with the 3x the M16 analysis allows (approximate).
Units in flight on that chip. It has no DRAM latency to cover: every one of its 1,024 cache reads per hash is an
on-die SRAM read. Its hash latency is the dependent chain: 128 items x (8 dependent SRAM reads plus 9 mixer
applications). At about 10 ns per on-die read and about 40 ns per 130-operation mixer on a 16-wide integer
pipeline at 2 GHz (both approximate), an item is about 0.4 us and a hash about 50 us; at 0.33 Ghash/s that is
about 17,000 hashes in flight, 530 units of 32 lanes. A tighter pipeline halves it. The GPU covers DRAM latency (40 to 48 ns row
cycle, MEMSYS 2018, more under load) with 130,560 lanes in flight at the measured occupancy (4,080 warps x 32), 348,160 at
full occupancy, that is 8 to 20 times more lanes than the chip needs.
What the scratch costs that chip, per variant, with the arithmetic:
Chip cache mirror: 128 mm^2, $46 per good die (N5 headline; 64 mm^2, $21 bit-cell lower bound). Chip scratch SRAM at
the same two densities (2.1 MB/mm^2 headline, 4.2 MB/mm^2 lower bound at N5):
| Variant | Dataset loads per hash | Chip ops per hash | Chip rate at 50 T op/s | 5090 rate | Chip gain | Chip scratch SRAM at 17,000 lanes, implicit store | Same, dense 64-slot store | Dense store as mm^2, headline / lower bound (N5) | Share of the 256 MiB mirror (any density) |
|---|---|---|---|---|---|---|---|---|---|
| scr0 (control), 128 loads | 128 | 150,000 | 333 MH/s | 139.7 measured | 2.4x | 0 | 0 | 0 | 0 |
| 12.5% replaced (scr2) | 112 | 131,400 | 381 | 160 projected (128/112 x 139.7) | 2.4x | 1.4 MB | 13 MB | 6.2 / 3.1 mm^2 | 4.9% |
| 25% replaced (scr4) | 96 | 112,800 | 443 | 186 projected | 2.4x | 2.7 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 50% replaced (scr8) | 64 | 75,600 | 661 | 279 projected | 2.4x | 5.4 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 12.5% added (16 RMW beside 128 loads) | 128 | 150,200 | 333 | 139.7 or below | 2.4x or more | 1.4 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 25% added | 128 | 150,400 | 332 | 139.7 or below | 2.4x or more | 2.7 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 50% added | 128 | 150,800 | 332 | 139.7 or below | 2.4x or more | 5.4 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 256-slot dense store (128 KiB class), any share | | | | | | | 53 MB | 25 / 12.6 | 20% |
How the rows are computed: a read-modify-write costs the chip about 12 integer operations (fold and rewrite) and
one SRAM access; replacing a load removes an item derivation (1,170 operations); the 5090's rate for a replaced
load is projected from the measured distinct-load bound (the card's rate tracks distinct dataset loads per hash,
`docs/bench-log.md` 3 October, 23.7 G loads/s at 1 GiB; the readwidth agent's M5 Max measurement of the night,
relayed by the coordinator, shows the same: 27.7 MH/s at v2 to 29.4-31.7 at 25 percent replaced and 44.4-49.1 at 50
percent, 32 KiB per warp). The scratch SRAM is 17,000 lanes x 80 to 320 bytes (implicit) or x 776 bytes (dense at
64 slots) or x 3,104 bytes (dense at 256 slots); its share of the mirror is a ratio of bytes, 4.9 or 20 percent,
whichever density is used for both; the implicit store (the chip's cheaper choice at every size, section 3.1) is
0.5 to 2 percent. The chip's gain is set by operations per dataset item and the GPU's distinct-load bound, and the
scratch touches neither.
Plain answer to the coordinator's question: no read-modify-write share under the 6 GB cap, replaced or added,
brings the named chip under 2x. The share would be chosen as the smallest at which the chip falls under 1.5x, and
there is none: the gain is 2.4x at 0, 12.5, 25 and 50 percent, 32 or 128 KiB. This changes nothing about the public
claim that layer 3 would have changed: the claim must rest on the mixer, not on the scratch.
The lever that does move that chip, from the M16 table, beside it:
| Mixer cost multiplier | Chip ops per hash | Chip rate | Gain against 139.7 MH/s, no fixed-function factor | With a 3x factor (approximate) | CPU verify per warp (M16 table, scaled from 0.41 to 1.2 ms) | 5090 daily dataset build |
|---|---|---|---|---|---|---|
| x1 (today) | 150,000 | 333 MH/s | 2.4x | 7.2x | 0.4 to 1.2 ms | 13.4 ms |
| x2 | 300,000 | 167 | 1.2x | 3.6x | 0.8 to 2.4 ms | 27 ms |
| x4 | 600,000 | 83 | 0.6x | 1.8x | 1.6 to 4.8 ms | 54 ms |
| x8 | 1,200,000 | 42 | 0.3x | 0.9x | 3.3 to 9.6 ms | 107 ms |
The mixer multiplier leaves the honest hash rate untouched (the miner pays the mixer once a day), costs the chip
linearly, and is bounded by the 10 ms verification gate (x8 is at the gate's edge on this core, and the 2019-class
core of O-1.14 is unmeasured). The scratch costs the honest GPU a measured share of its rate when it spills the
cache and nothing when it does not, and costs the chip a few megabytes. The comparison is not close.
### 3.5 Where the GPU's writes would cost DRAM latency, and why that does not help
The GPU's hot scratch footprint is not the nominal size either: it is the slots in-flight units have touched,
about `warps x 32 lanes x distinct slots x 16 bytes` (x 2 at a 32-byte sector, approximate): on the 5090 at 4,080
resident warps and scr4, 25.3 slots at 64 or 30.1 at 256, 53 to 63 MB of slots, 100 to 125 MB in sectors, around
the card's 96 MiB L2 (`docs/bench-log.md`, 3 October). The readwidth agent's M5 Max rows (coordinator's message:
the rate rises with the share at 32 and 128 KiB) show the scratch sitting in that chip's caches at 4,096 warps.
To push the writes to DRAM latency the hot footprint must pass the last-level cache at the resident count:
`96 MiB / 4,080 warps = 24 KiB per warp`, which at 512 bytes of touched slots per lane per read-modify-write slot
means `8k x 512 B > 24 KiB`, k above 6 (above 48 read-modify-writes per hash) at ANY nominal size on the table, or
a higher resident count. That fits the 6 GB cap (it is the hot set, not the arena, that matters), and it costs
the honest miner a DRAM-latency read-modify-write per slot (a DRAM row cycle is 40 to 48 ns across DDR4, GDDR5 and
HBM2, Li, Reddy and Jacob, MEMSYS 2018; the loaded latency a GPU kernel sees is higher, approximate; DRAM latency
improved 1.3x in two decades while bandwidth improved 20x, Chang 2017, so no memory technology an attacker could
buy removes it, and no shipped mining chip has used HBM or stacked memory) while the named chip still keeps the same
hot set in a few megabytes of SRAM at 8 to 20 times fewer lanes in flight. The write path cannot be made to cost the
chip more than the GPU, because the GPU must keep 8 to 20 times more of it live.
## 4. Question 3: the verifier's one-warp simulation is exact
### 4.1 Lazy fill on both sides
The CPU initialises lazily with a written bit per (lane, slot), one model per unit. The GPU initialises lazily with
a 32-bit tag in word 0 of each 16-byte slot: a slot whose tag equals the unit's tag reads as written, any other
reads as the fill (`emit.rs:143-157`). There is no explicit fill and no reset between units of a persistent warp
(`emit.rs:159-175`: the loop over `g_` keeps the arena). The two agree if and only if, when a unit first touches a
slot, that slot does not already carry the unit's tag. That is:
1. Tags are unique over the life of the arena's contents (`tag = salt + g_`, `salt` the host's running counter).
2. The arena holds no word equal to a live tag in a slot's tag position before the unit writes it.
### 4.2 The attacks (the bug classes)
| Case | What happens | Today |
|---|---|---|
| Recycled allocation | A fresh process starts `salt` at 1 (`packbench.swift:144`, `host.c:1027`). If the driver hands back the previous process's arena with its contents (Metal, CUDA and OpenCL do not promise zeroed memory, approximate), slots tagged 1..N from the old run match the new run's first units exactly, and those units read stale words instead of the fill: a CPU mismatch on every colliding slot. | not guarded; passes on this Mac because fresh allocations read as zero in practice and tag 0 is never issued (luck, not contract) |
| Tag counter wrap | `salt` is 32 bits and advances by units per launch. A 5090 at 139.7 MH/s runs 4.37 M units/s, 2^32 units in 984 s: the counter wraps every 16.4 minutes on one card (81.8 minutes on the M5 Max at 28 MH/s). After the wrap a slot whose LAST writer carried the repeated tag reads as written. With 10,880 arenas each slot is rewritten about 395,000 times between two uses of one tag (at 64 slots a unit leaves a slot untouched with probability 0.60; 0.60^395,000 is 0), so on a full card the wrap is harmless in practice; on a one-warp launch repeated 2^32 times it is not. | not guarded |
| Tag 0 on zeroed memory | A host that starts `salt` at 0 gives unit 0 the tag 0, which a zeroed arena carries in every slot: unit 0 reads zeros for every first touch. | both hosts start at 1; nothing in the pack says they must |
| `groups` not a multiple of the warp count | Warps run different trip counts; the OpenCL local-memory exchange path carries a barrier inside the loop (spec 1.9), so a short warp hangs or desynchronises. | `packbench` refuses it; `host.c` rounds the batch |
### 4.3 The host contract that makes the simulation exact
A host of a scratch class MUST: allocate the arena as `warps x 32 x words_per_lane` words and zero it; issue tags
from a 32-bit counter that starts at 1 and advances by the unit count of every launch; zero the arena again before
any launch whose tags would pass 2^32 - 1 (tag 0 is never issued); launch `groups` as a multiple of the warp count.
The zeroing costs one memset of the arena (340 MiB at 32 KiB x 10,880 warps) every 2^32 units, 16 minutes on a
5090. This is the class fix for all four rows: with it the GPU's tag test and the CPU's written bit are the same
predicate.
### 4.4 The tests (Metal, M5 Max, 5 October 2026)
Two consecutive units on one persistent warp and the wrap inside a launch (`packbench --warps 1`,
`--batch-base 4294967040`, the option added on this branch); the hand-built edge programs that force every
read-modify-write of a hash onto one slot (so two consecutive units on one arena collide on every slot); the
deliberate breaks. Results in section 7.2. On the CPU, the same edge programs against an independent hand model
(a second interpreter with its own slot store, `tests/scratch.rs`): 56 of 56 cases match, and the hand model with
its rewrite words swapped mismatches on every case (the comparison has teeth).
## 5. Question 4: the attack surface of the writes
| Surface | Argument | Test |
|---|---|---|
| Out of bounds | `s_ = rN & (slots - 1)`, so `s_ < slots`; the lane's arena is `(warp_ x 32 + lane) x 4 x slots` words from the base, the access is `arena + 4 x s_ + 0..3`, the largest index is `warps x 32 x 4 x slots - 1`, the host's allocation. The emitter has one scratch template (`emit.rs:143`) and it masks. | `scr_packs_regenerate_and_pass_the_static_scratch_check`: 42 of 42 emitted kernels (7 scr packs x 6 files, the OpenCL bound file carrying two kernels) regenerate byte for byte from program.json and pass the text check: k masked slot definitions with the class mask, k tagged stores, 3k fill calls, one arena definition with the class stride, one tag definition, no `scratch[`; six deliberate breaks caught (section 8) |
| Aliasing between lanes | Lane-major: lane l of warp w owns words `[(32w + l) x 4S, (32w + l + 1) x 4S)`; two (w, l) pairs give disjoint ranges. Inside the range a slot is 4 words at `4 x s_`, so two slots of one lane are disjoint too. | the `lanevar` edge program: one init-dependent slot per lane, 32 lanes at 64 slots share slots in pairs by the birthday bound; any cross-lane aliasing would change the fold; 128 of 128 lanes on Metal (section 7.2) |
| Wave64 (two logical units in one hardware wave) | `warp_ = tid >> 5`, so the two halves get `warp_ = 2w` and `2w + 1`, two arenas; `gbase` and `tag` are per `g_`, per half. | not run on wave64 hardware (the OpenCL emulator's persistent launch is on the readwidth commit; unverified here) |
| Determinism: alignment | A slot is 16 bytes at byte offset `16 x (lane_base + s_)`; the arena base is the buffer base: Metal, CUDA and OpenCL allocations are at least 128-byte aligned (CUDA 256, OpenCL `CL_DEVICE_MEM_BASE_ADDR_ALIGN` at least the largest built-in type, approximate from memory), so every 16-byte vector access is aligned. | Metal: every run of section 7 |
| Determinism: ordering | A lane's two read-modify-writes of the same slot in one hash are a load and a store, then a load and a store, from one thread to one address: program order within a thread holds in every model. No other thread touches the slot (aliasing row), so no atomics, fences or barriers are needed and none are emitted. | `slot0` and `sixteen` edge programs: 64 and 128 dependent read-modify-writes on one slot per lane per hash, standalone and as the second unit on a warp |
| Determinism: vendors | The statement is integer only: xor, rotate by immediate, multiply, add, a 16-byte load and store. Bit-exact across Metal, CUDA and OpenCL by construction; measured only on Metal here. | Metal; CUDA and OpenCL runs are PC jobs (not mine tonight) |
| 32-bit nonce wrap | `gbase = baseNonce + g_ x 32` and `nonce = baseNonce + gid` wrap in 32-bit arithmetic; `scr_fill(gbase + lane)` wraps like the CPU's `base.wrapping_add(lane)`; `out[gid]` indexes by launch position, not by nonce. An aligned unit never straddles 2^32 (spec 1.9), so the wrap case is a launch whose unit SEQUENCE crosses it. | `packbench --batch-base 4294967040 --batch-log2 9`: 16 units from 0xffffff00, the ninth at gbase 0; fingerprint identical at 1 and 4 warps (section 7.2); every fuzz pack runs that launch |
## 6. Question 5: what a vector for the scratch class must carry
Before a scratch pack can be a conformance vector (plan step 4, "only then a vector"), it must carry, beyond what
`igneum-program-pack-3` carries today:
1. The class in the program id and the pack (`scr<k>k<kb>`: it is, `program_id_class`, `generator.rs:400-412`)
and the geometry (slots per lane, words per lane, bytes per warp: it is, `program.h`).
2. The fill and the rewrite as text (it is, `program.json` "scratch").
3. The host contract of section 4.3 as text in `program.h` and `program.json`: tag counter from 1, zero at
allocation and at the wrap, `groups` a multiple of the warp count. Not there today.
4. Vectors that exercise the tag path, which the three standard units do not reliably: two consecutive units on
one warp (bases 0 and 32 in one one-warp launch) for a program whose consecutive units collide on a slot. At
64 slots any generated program collides (25 touched of 64 per unit; the broken-tag run of section 8 was caught by
base 1,000,000, a warp's 16th unit, and NOT by a two-unit launch whose vectors lack base 32). At 2,048 slots two
consecutive units share a touched slot with probability about 0.4 (32 x 32 / 2,048 expected overlaps = 0.5), so
the standard vectors would miss a broken tag at the 1 MiB size with probability about 0.6 per unit pair. The
edge programs `slot0` and `sixteen` collide on every slot at every size: a vector set should carry one.
5. A unit in the top 256 nonces with the launch crossing 2^32 (`--batch-base` near the top, at least two warps).
6. The batch fingerprint declared independent of the warp count (`8c07620f4d9adefd` for scr4k32 at 2^12 nonces
from base 0 at 1, 2 and 128 warps, section 7.2): unit independence is the property the per-unit reset gives, and
a fingerprint that moved with the warp count would mean a unit read another unit's slot.
## 7. Tests and results
### 7.1 CPU (`igneum-pow/tests/scratch.rs`, `cargo test -j4 --test scratch`, M5 Max, 5 October 2026, 3.6 s)
| Test | What | Result |
|---|---|---|
| `rewrite_is_a_bijection_of_the_fold_value` | 16 random slot contents x 2^16 consecutive fold values, each written word distinct; the three inversions | pass |
| `fill_is_a_bijection_of_the_nonce` | 7 slots x 3 words x 2^16 nonces distinct; lane and base interchange; wrap | pass |
| `written_words_unbiased_and_rehit_rates` | 6 classes x 3 seeds x 2^11 units, every rewrite traced: bias within 6 sigma (worst 3.63), re-hit rate within 0.9x to 2x of birthday, slot histogram, depth histogram | pass (tables of sections 2.3 and 3.2) |
| `edge_programs_match_the_hand_model` | 7 edge programs x 2 geometries x 4 bases (0, 32, 0x7ffffff0, 0xffffffe0) against an independent hand model; the slots driven and the re-hit counts as built; the mutated hand model mismatches | 56 of 56 pass, 56 of 56 teeth |
| `scr_packs_regenerate_and_pass_the_static_scratch_check` | 7 scr packs: program and program id from program.json, 6 kernel texts byte for byte, static scratch check on all 42, the pack's vectors from the CPU; six deliberate breaks caught | pass |
| `fuzz_scr_programs_cpu` | 200 generated programs over the six classes, generator contract and acceptance on every one, 4 units each (one in 0..224, one around 2^31, one in the top 256 nonces, one uniform), traced run equal to the untraced run, every slot inside the lane; writes the 214 packs for Metal with `IGNEUM_SCRATCH_PACKS_OUT` | pass; 200 of 200 have a unit in the top 256 |
| `slot_is_replayable_from_its_fold_values` | 64 dependent read-modify-writes replayed from the fill and the fold values; one dropped diverges | pass |
The rest of the crate: 33 of 34 lib tests and all pack tests pass; `verify::tests::fold_and_wide_fetch` fails on the
readwidth tip itself (`verify.rs:508`, `k as u32 * 0x9E37_79B1` overflows under the test profile's overflow
checks; the readwidth agent's test, reported to its owner, not touched here).
### 7.2 Metal (`proto-metal/packbench` built from this branch, M5 Max, 5 October 2026, under `with-lock.sh run`)
| Run | Launch | Expected | Result |
|---|---|---|---|
| scr4k32, standard pack | 2,048 warps, 2^24 nonces, 1 batch | 3 of 3 standalone, 3 of 3 in batch | PASS, fingerprint `3d1af881bd978fb9`; 1.8 s wall for the whole run (compile, cache, 1 GiB build, vectors, batch) |
| scr4k32, warp-count independence | 2^12 nonces (128 units) at 1, 2 and 128 warps | one fingerprint | `8c07620f4d9adefd` at all three, PASS |
| scr4k32, wrap inside the launch | 512 nonces from 0xffffff00 at 1 and 4 warps | one fingerprint, the base-0 vector inside the window after the wrap | `8e9e233234d3a297` at both, in-batch 1 of 1, PASS |
| scr4k32, broken tag (`tag = salt`), standard vectors | 2,048 warps, 2^24 | the base-1,000,000 vector (warp 530's 16th unit) fails | standalone 3 of 3, in batch 2 of 3, overall FAIL (caught) |
| scr4k32, broken tag, two units on one warp | 1 warp, 2^6 | nothing to catch it: the standard vectors have no base 32 | standalone 3 of 3, in batch 1 of 1, PASS (missed: the point of section 6 item 4) |
| 14 edge packs (7 programs x 32 and 128 KiB), run A | 1 warp, 2^6 (units at bases 0 and 32 on one arena) | 4 of 4 standalone, 2 of 2 in batch each | 14 of 14 PASS (56 of 56 standalone units, 28 of 28 in batch) |
| 14 edge packs, run B | 1 warp, 2^9 from 0xffffff00 (16 units on one arena, the wrap inside) | 4 of 4 standalone, 3 of 3 in batch each | 14 of 14 PASS (56 of 56, 42 of 42) |
| edge `slot0` at 32 and 128 KiB, broken tag (`tag = salt`) | 1 warp, 2^6 | standalone 4 of 4, in batch 1 of 2, FAIL | as expected at both geometries: the second unit read the first's slot 0 and FAILED; the standalone units passed |
| edge `slot0` at 32 KiB, broken lazy fill (`m_` forced to all ones: a first touch reads the stale words) | 1 warp, 2^6 | standalone fails | 0 of 4 standalone, 0 of 2 in batch, FAIL (lane 0 of base 0: GPU `64b49aeb987dae69`, expected `9ff3a2021f66b5be`) |
| 200 fuzz packs (scr2k32 29, scr4k32 26, scr8k32 42, scr2k128 32, scr4k128 26, scr8k128 45; datasets 64 MiB, 256 MiB, 1 GiB) | 2 warps, 2^9 from 0xffffff00 (8 units per warp, the wrap inside) | 4 of 4 standalone, 2 of 2 in the window, 200 of 200 PASS | 200 of 200 PASS: 800 of 800 standalone units (25,600 hashes), 400 of 400 in batch; 91 s for the 200 runs |
Totals on Metal: 228 of 228 runs PASS where a pass was expected, 3 of 3 FAIL where a failure was built in.
## 8. Deliberate breaks (the watcher rule)
| Break | Where | Caught by | Evidence |
|---|---|---|---|
| One slot mask dropped (Metal) | copy of scr4k32 `program.metal` | static check: "masked slot followed by the load: 3, expected 4" | test output |
| Mask 63 changed to 127 on every RMW (Metal) | same | "masked slot followed by the load: 0, expected 4" | test output |
| Arena stride 256 changed to 128 words (Metal) | same | "arena definition: 0, expected 1" | test output |
| A stray `arena[0]` and `scratch[1]` access (Metal) | same | "arena mentions: 10, expected 9; direct scratch indexing: 1, expected 0" | test output |
| One slot mask dropped (OpenCL, CUDA) | copies of scr4k32 `kernel.cl`, `kernel.cu` | "masked slot followed by the load: 3, expected 4" | test output |
| Wrong class geometry or RMW count or kernel count passed against a right text | the same text | the check fails | test output |
| `tag = salt` (every unit of a launch shares the tag) | copy of scr4k32 `program.metal`, on the GPU | the base-1,000,000 vector in a 2,048-warp batch | `vectors standalone 3/3, in batch 2/3`, overall FAIL |
| the same on the `slot0` edge pack, two units on one warp | on the GPU | in-batch 1 of 2 | bench-log entry |
| lazy fill broken (`m_` all ones) | copy of the `slot0` edge pack, on the GPU | standalone vectors | bench-log entry |
| The hand model's rewrite words swapped | `tests/scratch.rs` | every edge case mismatches | 56 of 56 |
The out-of-bounds break (mask dropped) was not run on the GPU on purpose: Metal does not bounds-check device
buffers (`TESTS.md` section 5), so a run would read another lane's or another buffer's words and "did not crash" would
prove nothing. The static check is the guard, as it is for the dataset mask.
## 9. What is unverified
1. CUDA and OpenCL runs of the scratch packs on NVIDIA and AMD (PC jobs, reserved for the readwidth agent tonight);
the 5090's rate per variant, so the "projected" column of section 3.4 is the distinct-load bound, not a
measurement. Wave64 hardware for the two-arena argument.
2. The chip-side latency figures of section 3.4 (10 ns SRAM read, 40 ns mixer) are approximate; the conclusion
does not depend on them: at ten times the in-flight count the scratch is still under a sixth of the mirror.
3. The recycled-allocation case was not reproduced (it needs a driver that hands back live contents); the argument
is that nothing forbids it and the contract of 4.3 removes it.
4. The slot-bias finding (section 3.2) was measured on three seeds per class; the hottest-slot ratio will vary by
program.
5. `verify::tests::fold_and_wide_fetch` on the readwidth tip (section 7.1).
## 10. Recommendation
1. Layer 3 is sound as a construct: the written words are uniform, the chain inside a unit has no short cut, the
kernels cannot write out of bounds, and with the host contract of section 4.3 the CPU's one-warp simulation is
exact (14 edge packs, 200 fuzz packs, the wrap, consecutive units on one arena, on Metal).
2. Layer 3 is not sound as a chip-resistance layer, at the capped size or at any size: CPU verification resets the
scratch per unit, so its live state is 8k slots per lane whatever the arena, a chip keeps it implicitly in 80 to
320 bytes per lane, and the named chip (on-die cache mirror plus recompute) keeps its whole scratch in 1.4 to
13 MB of SRAM at 530 units in flight, 3 to 5 percent of its mirror. Its gain stays at 2.4x (7x with a 3x
fixed-function factor, approximate) at 0, 12.5, 25 and 50 percent, replaced or added, 32 or 128 KiB. No share
under the 6 GB cap brings it under 2x.
3. Taking the read-modify-writes from the 16 dataset loads makes the hash less memory-hard for everyone: the GPU
measured faster at every share on the M5 Max (readwidth rows), and the chip's operations per hash fall with the
loads. If a scratch is kept at all, add it beside the 128 loads. There is no reason found here to keep one.
4. The lever that moves the named chip is the M16 mixer multiplier: x2 to 1.2x, x4 to 0.6x against the measured
5090 rate, at 0.8 to 4.8 ms of verification per warp against the 10 ms gate. Decision 2 should price that
against the gate on the 2019-class core (O-1.14) rather than layer 3.
5. If Josh keeps layer 3 for a reason outside this analysis: ship the host contract in the pack, add the four
vector items of section 6 (consecutive units with a forced collision, the wrap launch, the warp-count-independent
fingerprint, the contract text), and run the CUDA and OpenCL twins of section 7.2 on the PCs before the class
becomes a genesis rule.

View file

@ -0,0 +1,283 @@
# Layer 6: the SRAM mirror of the cache against published SRAM density, year 0 to 10
5 October 2026 (night), Counter ASIC 2.0 (`docs/plans/counter-asic-2.md`, layer 6), branch `ca2-analysis`. Every figure
below is either cited (paper, vendor document, URL, date) or labelled approximate. Nothing here is a measurement of a
chip. Numbers in this file were computed with the arithmetic shown; the script is in section 10.
Revision 2 (same night): the first draft priced the mirror from bit-cell area times a 0.70 array factor. The
coordinator's chip-economics research (sources below) showed that shipped cache-only dies land at about half that
density once assist circuits, redundancy, TSVs, power and test are in. Every table now carries two columns: the
shipped-product density as the headline and the bit-cell figure as the lower bound. The conclusion did not move; the
cost per die rose 2 to 3x.
## 1. The question
The lottery hash derives every dataset item from a 256 MiB cache (spec 01 sections 1.5 and 1.8). A chip that holds the
cache in on-die SRAM can recompute items instead of reading the dataset (ledger M16, the recompute attacker). Layer 6
asks whether the cache size, as the specification schedules it, keeps that SRAM mirror unaffordable for ten years of
the genesis schedule, and if not what growth rule would.
Two things also sit in a chip's SRAM budget if it mirrors the full read-only working set: the layer 5 hot table (32,
64 or 96 MB, a class parameter on `readwidth` b970dda, coordinator's note of 5 October) beside the 256 MiB cache. The
per-warp scratch of layer 3 (32 or 128 KB per warp, written, not read-only) is not mirrorable and is left out of the
mirror; it is counted in the 6 GB working-set budget in section 7.
## 2. What the specification schedules for the cache
| Quantity | Rule | Where |
|---|---|---|
| Dataset | 2 GiB at genesis plus 0.5 GiB per year (`N_d` grows about 23 KiB per day) | spec 01 section 1.13.3, Designed |
| Cache | 256 MiB, "prototype value, to be fixed at gate 1"; the rule that fixes it: "the cache must exceed the largest on-chip cache of any card that mines, and 96 MiB of L2 on the 5090 is the figure to beat" | spec 01 sections 1.5 and 1.16 |
| Cache growth | None. No section of `docs/spec/` grows the cache (grep of `docs/spec` for cache growth, schedule, doubling: only the dataset rule of 1.13.3 and the README's "growth" word, which refers to it) | this analysis, 5 October 2026 |
So the plan's layer 6 row ("already in the design; confirm the schedule") is half right: dataset growth is in the
design, cache growth is not. The cache is flat at 256 MiB for every year of the schedule as the spec stands. M16's
closing line names the rule the cache should get ("exceeds what one die can hold, and grows") as a gate 1 decision
that has not been taken.
## 3. SRAM density, cited: bit cells per node and shipped cache dies
### 3.1 Bit cells
| Node (vendor) | HD 6T bit cell, um^2 | Raw density, Mbit/mm^2 (1/cell) | Year of volume (approximate) | Source |
|---|---|---|---|---|
| N7 (TSMC) | 0.027 | 37.0 | 2018 | WikiChip, "TSMC Details 5 nm" (ISSCC/IEDM disclosures), https://fuse.wikichip.org/news/3398/tsmc-details-5-nm/ |
| N5 (TSMC) | 0.021 | 47.6 | 2020 | same (two N5 cells: HD 0.021, HP 0.025) |
| N3B (TSMC) | 0.0199 | 50.3 | 2022 to 2023 | WikiChip, "IEDM 2022: Did We Just Witness The Death Of SRAM?", https://fuse.wikichip.org/news/7343/iedm-2022-did-we-just-witness-the-death-of-sram/ (TSMC's IEDM 2022 N3 paper) |
| N3E (TSMC) | 0.021 | 47.6 | 2023 | same; Tom's Hardware, "TSMC's 3nm Node: No SRAM Scaling", https://www.tomshardware.com/news/no-sram-scaling-implies-on-more-expensive-cpus-and-gpus |
| N2 (TSMC) | 0.0175 | 57.1 | 2025 to 2026 | TSMC at IEDM 2024, reported by Tom's Hardware, https://www.tomshardware.com/tech-industry/tsmc-shares-deep-dive-details-about-its-cutting-edge-2nm-process-node-at-iedm-2024-35-percent-less-power-or-15-percent-more-performance ; ISSCC 2025 paper "A 38.1Mb/mm2 SRAM in a 2nm-CMOS-Nanosheet Technology", https://research.tsmc.com/page/memory/4.html |
| Intel 18A | 0.021 | 47.6 | 2025 to 2026 | ISSCC 2025 paper 29.2, "A 0.021 um^2 High-Density SRAM in Intel 18A RibbonFET Technology with PowerVia", https://www.researchgate.net/publication/389644177 ; IEEE Spectrum 26 Feb 2025, https://spectrum.ieee.org/sram-intel-tsmc |
| Samsung SF3 / SF2 | not disclosed as a bit cell area in anything found tonight (Samsung's ISSCC papers give assist circuits and macro figures, not the HD cell) | | | search of ISSCC 2021 to 2025 coverage, 5 October 2026; left out of the tables |
The stall. N3B's cell is 5% smaller than N5's and N3E's is the same size as N5's (0.021 um^2 both): zero SRAM
scaling from N5 to N3E (WikiChip IEDM 2022 article above; Tom's Hardware above; SemiAnalysis "TSMC's 3nm Conundrum",
https://newsletter.semianalysis.com/p/tsmcs-3nm-conundrum-does-it-even). N2's nanosheet cell recovers 17% (0.021 to
0.0175 um^2). So across 2020 to 2026 the HD bit cell shrank once, by 17%.
Macro density from the bit cell. WikiChip's and SemiAnalysis's convention is bit-cell density times about 0.70 for
the assist and periphery overhead (SemiAnalysis, December 2022: TSMC N5 HD SRAM macro 31.8 Mib/mm^2 after about 30%
assist overhead; WikiChip's 31.8 Mib/mm^2 for the 0.021 um^2 cell is the same arithmetic). The two ISSCC 2025 macros
bracket it: TSMC N2 38.1 Mb/mm^2 at a 0.0175 um^2 cell is 67%; Intel 18A 38.1 Mb/mm^2 array density and 34.3 Mb/mm^2
for the volume macro at a 0.021 um^2 cell are 80% and 72%. That is a macro on a test chip. It is the LOWER BOUND on
die area, not the die.
### 3.2 Shipped cache dies (what a whole die of SRAM really holds)
| Product | SRAM | Die | Node | MB per mm^2 | Source |
|---|---|---|---|---|---|
| AMD 3D V-Cache (Zen 3 SRAM chiplet) | 64 MB | 41 mm^2 | TSMC 7 nm | 1.56 | AMD at Hot Chips 33, reported by Tom's Hardware, August 2021, https://www.tomshardware.com/news/amd-unveils-more-ryzen-3d-packaging-and-v-cache-details-at-hot-chips ("the 3D V-Cache SRAM measures 41 mm^2", "64 MB of 7 nm SRAM"); the densest cache-only die that has shipped |
| Graphcore GC200 (with compute) | 900 MB | 823 mm^2 | 7 nm | 1.09 | coordinator's chip-economics research, 5 October 2026 (vendor figures) |
| Groq TSP | 220 MB | 725 mm^2 | 14 nm | 0.30 | same |
The V-Cache die is a pure SRAM die with its TSVs, redundancy, test and power: 1.56 MB/mm^2 at N7 against the bit-cell
figure 37.0 Mbit/mm^2 = 4.6 MB/mm^2 and the 0.70-macro figure 3.2 MB/mm^2. The shipped die is 0.48 of the macro
figure. The headline column below scales the V-Cache density to other nodes by the bit-cell ratio (0.027 / cell), an
approximation that assumes the periphery and TSV overheads scale with the cell, which they do not fully (so the
headline column is itself slightly optimistic for the attacker at N5 and below).
### 3.3 GPU on-die SRAM, the reticle, wafer prices
GPU on-die SRAM for scale: the RTX 5090 carries 96 MB of L2 (98,304 KB) on a 750 mm^2 TSMC 4N die with 92.2 billion
transistors; the full GB202 has 128 MB; the RTX 4090 had 72 MB and the RTX 3090 6 MB (NVIDIA, "RTX Blackwell GPU
Architecture" whitepaper v1.1, appendix table "L2 Cache Size", https://images.nvidia.com/aem-dam/Solutions/geforce/blackwell/nvidia-rtx-blackwell-gpu-architecture.pdf).
At the V-Cache density scaled to N5 (2.0 MB/mm^2) that L2 is about 48 mm^2 of the 750 (6%), approximate. The
RX 9070 XT carries 64 MB of Infinity Cache plus 8 MB of L2 (vendor figures, approximate, bench-log "the 9070 XT on the
eGPU").
Reticle: the EUV field is 26 x 33 mm = 858 mm^2, about 830 mm^2 usable after scribe lanes (SemiAnalysis, "Die Size
And Reticle Conundrum", https://newsletter.semianalysis.com/p/die-size-and-reticle-conundrum-cost ; WikiChip "Mask",
https://en.wikichip.org/wiki/mask). The 5090's 750 mm^2 is 90% of it.
Wafer prices (approximate; TSMC publishes none, every figure is supply-chain reporting): N7 about $9,500, N5 and N3
about $20,000 (Silicon Analysts, "Wafer Pricing by Node", September 2026, https://siliconanalysts.com/data/wafer-pricing);
N2 about $30,000 (Tom's Hardware, https://www.tomshardware.com/tech-industry/semiconductors/tsmc-could-charge-up-to-usd45-000-for-1-6nm-wafers-rumors-allege-a-50-percent-increase-in-pricing-over-prior-gen-wafers).
## 4. Die area to mirror the cache, per node, two columns
Headline = V-Cache density (41 mm^2 per 64 MiB at N7) scaled by the bit-cell ratio. Lower bound = bits / (raw
density x 0.70). Columns: the 256 MiB cache alone, the cache plus the 96 MB hot table of layer 5 (as MiB), and the
larger caches of the options in section 7. Area in mm^2; a figure over 830 is split into the dies shown.
| Node | 256 MiB, headline | 256 MiB, lower bound | 256 + 96, headline | 256 + 96, lower bound | 512 MiB, headline / lower | 1 GiB, headline / lower | 4 GiB, headline / lower |
|---|---|---|---|---|---|---|---|
| N7 | 164 | 83 | 226 | 114 | 328 / 166 | 656 / 331 | 2,624 (4 dies) / 1,325 (2 dies) |
| N5 | 128 | 64 | 175 | 89 | 255 / 129 | 510 / 258 | 2,041 (3 dies) / 1,031 (2 dies) |
| N3B | 121 | 61 | 166 | 84 | 242 / 122 | 483 / 244 | 1,934 (3 dies) / 977 (2 dies) |
| N3E, Intel 18A | 128 | 64 | 175 | 89 | 255 / 129 | 510 / 258 | 2,041 (3 dies) / 1,031 (2 dies) |
| N2 | 106 | 54 | 146 | 74 | 213 / 107 | 425 / 215 | 1,701 (3 dies) / 859 (2 dies) |
One reticle (830 mm^2) holds, at the headline density, 1.3 GiB of SRAM at N7, 1.6 GiB at N5, N3E and 18A, 1.9 GiB at
N2 (lower-bound column: 2.5, 3.2, 3.9 GiB).
Against the figures the ledger carries: M16's "100 to 300 mm^2" (low end from a 0.02 um^2 cell with overhead, high
end from wafer-scale parts at about 1 MB per mm^2) brackets the headline 106 to 164 mm^2 well; the plan's "about
45 mm^2 at a leading node" is below even the lower bound and should be read as the bit-cell area with no overhead.
The right figures for the ledger are 106 to 164 mm^2 (shipped density) with 54 to 83 mm^2 as the floor.
## 5. Cost per good die, two columns
Dies per 300 mm wafer by the usual approximation pi x 150^2 / A minus the edge term pi x 300 / sqrt(2A); yield by
Poisson exp(-A x D0) with D0 = 0.1 defects per cm^2 (an assumption, approximate; SRAM arrays carry redundancy so
real yield is higher, which lowers these costs). Cost per good die = wafer price / (dies x yield). Packaging, test,
the logic beside the SRAM and the design (masks at N5 and below run into the tens of millions of dollars,
approximate) are not in these numbers; they are per-die silicon only. Headline / lower bound in each cell.
| Node, wafer price | 256 MiB | 256 + 96 MiB | 1 GiB | 4 GiB |
|---|---|---|---|---|
| N7, $9,500 | 164 mm^2, 379 dies, yield 0.85: $30 / $13 | $44 / $19 | $224 / $75 | $896 (4 dies) / $456 (2 dies) |
| N5, $20,000 | 128 mm^2, 495 dies, 0.88: $46 / $21 | $68 / $30 | $306 / $111 | $1,512 (3 dies) / $621 (2 dies) |
| N3B, $20,000 | 121 mm^2, 524 dies, 0.89: $43 / $20 | $63 / $28 | $280 / $103 | $1,371 (3 dies) / $569 (2 dies) |
| N3E, 18A, $20,000 | $46 / $21 | $68 / $30 | $306 / $111 | $1,512 / $621 |
| N2, $30,000 | 106 mm^2, 600 dies, 0.90: $56 / $26 | $81 / $37 | $343 / $131 | $1,641 (3 dies) / $696 (2 dies) |
Reading. The silicon for a 256 MiB mirror is $30 to $56 per die at shipped density (2 to 3x the first draft's
figure), under $90 with the hot table. A funded chip programme pays that without noticing: it was never the SRAM
that priced the recompute attacker out, and the plan's premise for layer 6 ("the SRAM mirror stays unaffordable")
does not hold for the cache as a mirror and did not hold at genesis either. A 1 GiB cache is a 425 to 656 mm^2 die
($224 to $343), affordable too; 4 GiB is a 3 to 4 die part at about $900 to $1,600 of silicon, which is a different
product but not an impossible one (the attacker's problem at that size is the 1,024 dependent cross-die reads per
hash, section 6).
## 6. What the mirror buys the attacker, year by year
From M16 (`docs/analysis/m16-recompute-attacker-2026-10-05.md`): with the cache on die the attacker recomputes 128
items per hash at about 1,170 integer operations and 8 dependent 64-byte cache reads each, about 150,000 operations
and 1,024 dependent SRAM reads per hash. At a 5090-class integer budget (about 50 T op/s, approximate) that is
0.33 Ghash/s against the honest 141 Mhash/s projected for version 2 programs: 2.4x at equal silicon before any
fixed-function factor, 3x to 6x with one (approximate). The SRAM is 106 to 164 mm^2 of that chip at the headline
density (14 to 22% of a 750 mm^2 die; the m16 model's 13 to 40% band holds), so the mirror is cheap and the recompute
route is bound by integer throughput, not by SRAM.
The layer 5 hot table changes nothing in that arithmetic: the hot table is read-only and derived from the day key
like the cache, so a chip mirrors it in the same SRAM (another 32 to 96 MB, 24 to 48 mm^2 at N5 headline) and reads
it at SRAM latency, which is exactly what a GPU's L2 does with it. Layer 5 taxes the DRAM-only chip (the one without
SRAM); it does not tax the SRAM chip.
Dataset growth does not touch the recompute attacker: the attacker never holds the dataset. It taxes the
partial-store attacker (O-1.6, the time-memory curve, not drawn) and the honest card.
Year by year under the schedule as it stands (flat 256 MiB), the mirror's area at the best node available that
year, headline density. Node years are approximate; the density trend from 2018 to 2025 is 37.0 to 57.1 Mbit/mm^2
raw, 1.54x in 7 years, about 6% per year, and it came in one step (N2); the extrapolation past 2026 assumes that
average holds (approximate, and optimistic for the attacker: A16 and A14 have no disclosed SRAM cell yet).
| Year | Calendar (approximate) | Dataset, GiB | Cache (spec) | Best node | Mirror of the cache, headline (lower bound), mm^2 | With a 96 MiB hot table, headline, mm^2 | Mirror as a share of a 750 mm^2 die |
|---|---|---|---|---|---|---|---|
| 0 | 2027 | 2.0 | 256 MiB | N2 (cited) | 106 (54) | 146 | 14% |
| 1 | 2028 | 2.5 | 256 MiB | N2 or A16 | 103 (52) | 142 | 14% |
| 2 | 2029 | 3.0 | 256 MiB | trend | 95 (48) | 130 | 13% |
| 3 | 2030 | 3.5 | 256 MiB | trend | 89 (45) | 123 | 12% |
| 4 | 2031 | 4.0 | 256 MiB | trend | 84 (43) | 116 | 11% |
| 5 | 2032 | 4.5 | 256 MiB | trend | 79 (40) | 109 | 11% |
| 6 | 2033 | 5.0 | 256 MiB | trend | 75 (38) | 103 | 10% |
| 7 | 2034 | 5.5 | 256 MiB | trend | 71 (36) | 97 | 9% |
| 8 | 2035 | 6.0 | 256 MiB | trend | 67 (34) | 92 | 9% |
| 9 | 2036 | 6.5 | 256 MiB | trend | 63 (32) | 87 | 8% |
| 10 | 2037 | 7.0 | 256 MiB | trend | 59 (30) | 82 | 8% |
Reading. A flat cache's mirror shrinks from 14% to 8% of a large die over the decade, and a 5090-class consumer GPU
already carries 96 MB of L2 on one die with the full GB202 at 128 MB; at the 2020 to 2025 pace of GPU L2 growth
(6 MB, 72 MB, 96 MB on the three NVIDIA flagships in the whitepaper table) a consumer GPU could hold 256 MiB on die
within the decade. The spec's own rule for the cache ("must exceed the largest on-chip cache of any card that
mines") would then be broken by a flat cache. That is the real reason to grow it: not to price a chip out (section
5 shows the SRAM cannot do that) but to keep the cache out of every GPU's own cache, so the honest hash stays
DRAM-latency-bound and the recompute route stays a route only a custom chip can take.
## 7. Answer to the layer 6 question, and the options
Does the flat 256 MiB cache keep the SRAM mirror unaffordable through year 10? No. It is affordable at year 0 ($30 to
$56 of silicon per die at shipped density, section 5) and gets cheaper. What keeps the recompute attacker near 1x is
M16's integer arithmetic and the mixer-cost lever (4x the mixer cost puts the equal-silicon gain at 0.36x, bounded
by the CPU verify gate), not the cache size. The cache size does one other job, keeping the cache larger than any
GPU's L2, and that job needs growth.
Options for the cache rule, with the honest costs each implies. Verifier fill time is 0.2 s per 256 MiB on one core
(spec 1.12: "a 0.2 s CPU cache fill", from the measured 175 to 190 ms of section 1.8.3), scaled linearly; the
verifier holds the whole cache (section 1.11), so its memory is the cache size plus the program and the interpreter.
GPU fill: 0.67 ms per 256 MiB on the 5090 (section 1.8.3), linear. The GPU dataset build (13.4 ms per 1 GiB on the
5090, section 1.8.3) depends on the dataset size, not the cache size; a larger cache spreads the build's 8 dependent
reads per item over more memory, which on a GPU means more of them miss L2 and the build slows by some factor
between 1x and the L2-to-DRAM latency ratio, which is a measurement to take (approximate; owed). Mirror area is at N2
headline density (lower bound in brackets), the node of the first years; at the trend's year-10 density divide by
about 1.8.
| Option | Rule | Cache at year 0 / 4 / 10 | Mirror at N2, headline (lower bound), year 0 / 4 / 10, mm^2 | Dies at year 10 (830 mm^2 reticle), headline | Verifier fill, one core, year 0 / 10 | Verifier memory, year 10 | GPU cache fill (5090), year 10 | Keeps the cache above a 96 MB L2 at year 10 | Keeps it above a 256 MB L2 |
|---|---|---|---|---|---|---|---|---|---|
| A, as specified | flat 256 MiB | 256 / 256 / 256 MiB | 106 (54) / 106 / 106 | 1 | 0.2 / 0.2 s | 256 MiB | 0.7 ms | yes, 2.7x | no |
| B | cache = dataset / 8 (today's ratio) | 256 / 512 / 896 MiB | 106 (54) / 213 (107) / 372 (188) | 1 | 0.2 / 0.7 s | 896 MiB | 2.3 ms | yes, 9.3x | yes, 3.5x |
| C | cache doubles when the dataset doubles (the dataset's own clock: year 4, then year 12) | 256 / 512 / 512 MiB | 106 (54) / 213 (107) / 213 (107) | 1 | 0.2 / 0.4 s | 512 MiB | 1.3 ms | yes, 5.3x | yes, 2x |
| D | cache = dataset / 4 | 512 / 1,024 / 1,792 MiB | 213 (107) / 425 (215) / 744 (376) | 1 | 0.4 / 1.4 s | 1.75 GiB | 4.7 ms | yes | yes, 7x |
| E, one reticle | cache sized so the mirror exceeds one reticle at the node of the day: 2 GiB at N2 headline density (section 4; 4 GiB on the lower bound), growing with density | 2 GiB / about 2.3 / about 3.5 GiB | 850 / 850 / 850 (by construction) | 2 | 1.6 / 2.8 s | 3.5 GiB | 5.4 / 9.4 ms | yes | yes |
Where the working set enters (coordinator's budget: 1 GiB table + hot table + scratch for every resident warp +
buffers under 6 GB on an 8 GB card): the cache is not in the miner's working set at hash time (the dataset is built
from it once a day and the cache can be dropped or kept), so options A to D do not move that budget; the dataset's own
growth does (2 GiB at genesis, 4 GiB at year 4, 7 GiB at year 10, which is past an 8 GB card at about year 8 on its
own). Option E's 2 GiB cache would have to be built on the card and dropped, which is fine for a 16 GB card and tight
on an 8 GB one at build time (2 GiB cache + 2 GiB dataset + hot table). The per-warp scratch at 170 SMs x 64 warps
(approximate, readwidth) is 340 MB at 32 KB and 1.36 GB at 128 KB per warp; with the 1 GiB table, a 96 MB hot table
and buffers that is 1.5 to 2.5 GB at the prototype dataset size, 2.5 to 3.5 GB at the 2 GiB genesis size, inside
6 GB either way.
Recommendation. Option C (the cache doubles when the dataset doubles) is the one that keeps the spec's own rule true
with the smallest verifier cost: it ties the cache to a clock the spec already has, keeps `AND MASK` (a power of two
every step, which is the 1.13.3 option (b) argument again), costs the verifier 0.4 s and 512 MiB at year 4 and nothing
more until year 12, and keeps the cache 2x above a 256 MB GPU L2 if one appears. It does not price a chip out; nothing
about cache size does (section 5). The lever that does is the mixer cost multiplier of M16, which is the gate 1
decision to take beside this one. Option B is the same idea in a smooth form and costs the verifier 0.7 s at year 10.
Option E is the only one that makes the mirror a multi-die part and it costs every verifier 1.6 s and 2 GiB at
genesis (at the headline density; the lower-bound density would ask for 4 GiB and 3.2 s), which fails the spirit of
the 10 ms verify gate (the fill is once a day, but a light node joining pays it on every day it syncs across).
Decision for Josh, at gate 1: A, B, C, D or E above, together with M16's mixer multiplier. Nothing here changes a
vector today: the cache size is a prototype value of spec 1.16 and the growth rule would be a new sentence in 1.13.3.
## 8. Why the latency bound is the property to lean on (citations behind the plan's rule)
The plan's "what stays true" paragraph says DRAM latency is the same physics for everyone and bandwidth per watt is
what a custom memory chip buys. The sources behind that:
| Claim | Figure | Source |
|---|---|---|
| Random-access DRAM latency is the same across memory types | Row cycle time 40 to 48 ns across DDR4, GDDR5 and HBM2 | Li, Reddy and Jacob, "A Performance and Power Comparison of Contemporary DRAM Architectures", MEMSYS 2018 (coordinator's chip-economics research, 5 October 2026) |
| Latency does not scale, bandwidth does | DRAM latency improved about 1.3x in two decades while bandwidth improved about 20x | K. Chang, "Understanding and Improving the Latency of DRAM-Based Memory Systems", PhD thesis, CMU, 2017 (same research) |
| No mining chip has bought latency with exotic memory | No shipped mining chip has used HBM or stacked memory; the Ethash chips used DDR3, GDDR6 and undisclosed types | same research; the Ethash chip gain of about 3x in the plan came from bandwidth per watt, not latency |
| The honest hash is latency-bound on every card measured | The hash runs within a few percent of 1/128 of each card's dependent random-read ceiling (5090, 9070 XT, M5 Max) | `docs/bench-log.md`, "the 9070 XT on the eGPU", 5 October 2026 (measured) |
Reading for layer 6: an SRAM mirror beats DRAM latency by about 10x per read (a 64 MiB buffer inside the 9070 XT's
Infinity Cache chased at 9.2 G loads/s against 2.5 in GDDR6, the same bench-log entry; the 5090's L2 at 5.8x the
hash rate of its 1 GiB dataset, M16), which is why the recompute attacker is bound by the 1,024 dependent SRAM reads
and the 150,000 integer operations per hash and not by the SRAM's size or price. The cache size decides whether the
mirror is one die or several (section 4); it does not decide whether the mirror exists.
## 9. What is cited, what is approximate, what is owed
| Item | Status |
|---|---|
| Bit cells for N7, N5, N3B, N3E, N2, Intel 18A | cited (section 3.1) |
| Shipped cache-die density (AMD V-Cache 64 MB on 41 mm^2 at 7 nm; Graphcore GC200; Groq TSP) | cited (section 3.2; V-Cache checked against Tom's Hardware's Hot Chips 33 report, 5 October 2026; the Graphcore and Groq rows are from the coordinator's research and were not re-checked tonight) |
| Scaling the V-Cache density to other nodes by the bit-cell ratio | approximate, stated |
| Samsung SF2 or SF3 bit cell | not found; left out |
| Array efficiency 0.70 | WikiChip's and SemiAnalysis's convention, bracketed by two ISSCC 2025 macros (67 to 80%); a macro figure, used only as the lower bound |
| Wafer prices | approximate, supply-chain reporting, cited |
| D0 = 0.1 per cm^2, Poisson yield | assumption, stated |
| Node years and the 6% per year density trend past 2026 | approximate, extrapolated from cited 2018 to 2025 points |
| GPU L2 sizes | cited (NVIDIA whitepaper); AMD Infinity Cache approximate |
| Latency citations (MEMSYS 2018, Chang 2017, mining-chip memory types) | from the coordinator's research, not re-read tonight |
| Recompute attacker arithmetic | M16, which is itself arithmetic on measured rates, not a chip measurement |
| Dataset-build slowdown at a larger cache on a GPU | owed, a measurement (5090 at a 512 MiB and 1 GiB cache) |
| The on-die emulation of M16 (inline kernel with a 64 MiB cache inside the 5090's L2) | still a PC job (M16) |
## 10. The arithmetic
```
MiB = 2^20; bits = cache_MiB * MiB * 8
headline_mm2 = cache_MiB * (41 / 64) * (cell_um2 / 0.027) (V-Cache: 41 mm2 per 64 MiB at N7, scaled by cell)
raw_Mbit_per_mm2 = 1 / cell_um2 (1e6 cells per mm2 per um2 of cell)
lower_bound_mm2 = bits / (raw * 0.70 * 1e6)
dies_per_wafer = pi * 150^2 / area - pi * 300 / sqrt(2 * area)
yield = exp(-area_mm2 * 0.001) (D0 = 0.1 per cm2)
cost_per_good_die = wafer_price / (dies * yield); over 830 mm2: k = ceil(area / 830) dies of area / k, cost x k
reticle_GiB = 830 / (mm2 per MiB) / 1024
```
Run on 5 October 2026 with Python 3 on the M5 Max; the printed tables are the ones above, rounded.

View file

@ -1551,6 +1551,148 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus
Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`.
## 5 October 2026 (night), read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, a written scratch; three cards (gate 1 experiment, cryptographer)
Branch `readwidth` (commits 019b014, b970dda, 4badcee, a9e002c, d0018cf and the entry commit); plan and recommendation in `docs/plans/read-width.md`. Nothing here changes consensus: every class sits behind `igneum-pow --class` and the default class is generator version 2 byte for byte (`igneum-pow/tests/packs.rs` passes on the four pinned packs after every commit). Question (Josh, after "the 9070 XT on the eGPU" above): would wider reads keep the latency-bound random-access property while closing the vendor gap. Additions from the coordinator: a per-load width drawn from an era-fixed mix, and a written per-warp scratch (measurement only, no soundness claim).
**What a class does** (`igneum-pow/src/generator.rs` `LoadClass`, `verify::fold_words`, the three emitters): a load of W words reads the W-word-aligned address `(src AND MASK) AND NOT (W - 1)` and folds every word into `dst` (`x = dst ^ w0; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]`); W = 1 is the lottery hash exactly (`w4` = pack `bcc1248b10cc90f2`). A mix class draws W per load with one extra `below(100)` roll per instruction. A scratch class `scr<k>k<kb>` turns `k` of the 16 memory slots into read-modify-writes of a 16-byte slot of the lane's share of a `kb` KiB per-warp scratch (kernels run persistent warps, one per block or work-group; a slot reads as a seed-and-base fill until the unit writes it, behind a per-unit tag). Program ids carry the class. Dependent chain and 32-lane unit unchanged.
**Correctness**: 23 packs (`proto-cuda/packs-readwidth/`, Rust CPU reference vectors). Every pack passed its three vector units and the cache and dataset checks on Metal (M5 Max, `proto-metal/packbench`), Apple OpenCL (`--bench-pack`), the RTX 5090 (NVRTC, `igneum-worker-cuda --bench`) and, the 16 width and mix packs, the RX 9070 XT (`igneum-worker-opencl --bench-pack`); the 2^24 batch fingerprints agree across all four runtimes on every pack (for example w16 `e7c890445b47af60`, w64 `836e56e7d496e980`, mixB-2 `a18ac73098c76007`). The clang CUDA emulation (w16, w64, w64x4, mixA-0, mixB-0: 3 of 3 units standalone and 2 of 2 in batch at 2 warps per block) and the clang OpenCL emulation (the same five plus scr2k32 and scr8k128, sub-group 32 and, width packs, wave64 with sub-group shuffles) pass with equal fingerprints per configuration. Acceptance rule on the classes: 60 candidates per class, rejection 0 to 14 of 60 (w16 and w64 as v2; the mixes the same; the scratch classes' distinct-address bound now covers dataset loads only, since a 64-slot lane scratch repeats slots by design). CPU verifier (M5 Max, one core, avg of 50 units, `igneum-pow bench --class`): v2 0.604 ms, w16 0.610, w64 0.630, w64x4 0.160, mix50-35-15 0.620, mix25-50-25 0.614, scr0k32 0.600 (1.004 on a loaded re-run), scr2k32 0.657, scr4k32 0.458, scr8k32 0.317, scr2k128 0.535, scr4k128 0.458, scr8k128 0.311; per hash divide by 32. The wide reads cost the verifier nothing (a lane's words lie in one item); scratch ops replace item derivations and make it cheaper.
**Probes** (`--memprobe`, dependent random reads at 1024 MiB, G reads/s, best over lanes in flight; 4 B = the hash's pattern; the 5090 and 9070 XT with the card off in the app, the Mac through Apple OpenCL under a load average of 5 to 10):
| Card | 4 B chase | 16 B | 64 B | 64 B as GB/s | coalesced stream GB/s (rated) | integer chain |
|---|---|---|---|---|---|---|
| RTX 5090 (PC 2, CUDA) | 17.5 to 18.2 | 18.0 to 19.9 | 9.1 to 15.7 (9.1 at 4 M lanes) | 584 | 1,579 (1,792) | 39.0 T op/s |
| RX 9070 XT (PC 1, eGPU, OpenCL) | 2.42 to 2.66 | 2.43 to 2.73 | 2.47 to 2.87 | 158 | 636 (640) | 6.2 T op/s |
| Apple M5 Max (Apple OpenCL, approximate) | 3.50 | 3.51 | 3.51 | 225 | 522 | |
Reading: on the 9070 XT and the M5 Max a 64-byte dependent read costs exactly what a 4-byte one costs (the line is fetched either way); on the 5090 a 64-byte read costs about two 4-byte reads (two 32-byte sectors) and the 64 B chase at full occupancy sits at 584 GB/s, a third of the stream.
**Hash rates** (5 timed dispatches of 2^24 nonces after a warm-up; Metal and the 9070 XT by device time, the 5090 by wall time around the stream sync; the PC cards switched off in the app for the run and restored, PC 1's 5090 and the integrated chip kept mining; the Mac under other agents' builds, load 4 to 9, so its absolute numbers carry that; the share = measured / (the card's probe ceiling at the class's widths / loads per hash)):
| Class | dataset B/hash | RTX 5090 MH/s (share) | RX 9070 XT MH/s (share) | M5 Max Metal MH/s (share) | 5090 / 9070 |
|---|---|---|---|---|---|
| v2 (w4, the lottery hash) | 512 | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x |
| w16 | 2,048 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x |
| w64 | 8,192 | 71.9 (0.58) | 17.59 (0.78) | 28.27 (1.03) | 4.1x |
| w64x4 (32 loads) | 2,048 | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x |
| mix50-35-15, 6 programs: min / median / max (spread of median) | 1,664 to 3,680 | 99.5 / 114.2 / 121.0 (18.8%) | 17.45 / 18.76 / 18.83 (7.4%) | 25.36 / 27.26 / 28.43 (11.3%) | 6.1x |
| mix25-50-25, 6 programs | 2,240 to 5,024 | 95.9 / 107.3 / 119.8 (22.3%) | 17.84 / 18.45 / 18.85 (5.5%) | 23.21 / 24.68 / 25.21 (8.1%) | 5.8x |
Scratch (variant 5; N persistent warps; 5090: 2,048 warps launched against a resident capacity of 4,080 = 24 blocks/SM x 1 warp/block x 170 SMs at `--block-warps 1`, the occupancy query unchanged by the allocation (24 before and after); Metal: 2,048 to 16,384 warps swept, best shown; arena = N x per-warp size; the whole working set = 1 GiB dataset + 256 MiB cache + 128 MiB output + arena, under 2 GB on every row):
| Class (k of 16 slots, KiB per warp) | scratch ops/hash | dataset B/hash | RTX 5090 MH/s (vs scr0, share) | M5 Max Metal MH/s (vs scr0) | RX 9070 XT MH/s | 5090 arena / working set |
|---|---|---|---|---|---|---|
| scr0k32 (control, persistent loop, no RMW) | 0 | 512 | 139.1 (0, 0.98) | 28.25 (0) | 17.88 (control, 0.86) | 64 MiB / 1.4 GiB |
| scr2k32 (12.5%) | 16 | 448 | 114.4 (-18%, 0.80) | 26.14 (-7%) | 14.65 (-18%) | 64 MiB / 1.4 GiB |
| scr4k32 (25%) | 32 | 384 | 109.8 (-21%, 0.76) | 31.74 (+12%) | 14.00 (-22%) | 64 MiB / 1.4 GiB |
| scr8k32 (50%) | 64 | 256 | 122.1 (-12%, 0.82) | 49.08 (+74%) | 14.17 (-21%) | 64 MiB / 1.4 GiB |
| scr2k128 (12.5%) | 16 | 448 | 110.1 (-21%, 0.77) | 26.24 (-7%) | 14.07 (-21%) | 256 MiB / 1.6 GiB |
| scr4k128 (25%) | 32 | 384 | 98.0 (-30%, 0.68) | 28.08 (-1%) | 13.14 (-27%) | 256 MiB / 1.6 GiB |
| scr8k128 (50%) | 64 | 256 | 72.8 (-48%, 0.49) | 35.44 (+25%) | 12.03 (-33%) | 256 MiB / 1.6 GiB |
The 9070 XT rows are 2,048 persistent warps (4,096 within 1 percent), arena 64 MiB at 32 KiB and 256 MiB at 128 KiB, working set 1.4 and 1.6 GiB; its control (17.88, the persistent loop) equals its v2 rate (18.15) within 2 percent, and every RMW share costs it 18 to 33 percent: on AMD a scratch op is a dependent 16-byte read plus a write into a region the 64 MB Infinity Cache does not hold for 2,048 warps, so it is memory work there as on the 5090, not the cached op it is on Apple. Apple OpenCL on the same scratch packs (wall time, `--bench-pack --warps 2048`): scr0k32 27.85, scr2k32 28.58, scr4k32 32.43, scr8k32 47.93, scr2k128 25.67, scr4k128 27.24, scr8k128 32.93 MH/s, the Metal shape within 4 percent, fingerprints equal. Bytes moved per scratch op: 16 read + 16 written (the tag word included); per hash at 50 percent, 1,024 read + 1,024 written beside 256 of dataset reads. The 5090 at 4,096 launched warps (above its 4,080 resident) lost 2 to 26 percent (scr8k32 90.0 MH/s), so the rows above are the in-capacity launch.
**Readings.** (1) Same count, wider: the vendor gap does not move at 16 B (7.8x) because on the 9070 XT a 4-byte read already costs a 64-byte line and on the 5090 a 16-byte read costs one 32-byte sector, the same as 4 bytes: the memory systems do identical work, only the fold's input grows. At 64 B the gap closes to 4.1x, entirely by the 5090 losing half its rate (its share falls to 0.58 and its DRAM traffic reaches 589 GB/s, 37 percent of the stream: bandwidth, not latency, bounds it), while the 9070 XT and the M5 Max do not move. (2) Fewer, wider (w64x4): 3.7x, but every card runs 4x faster because the dependent chain is 32 loads long instead of 128; the 5090 sits at a 0.56 share (bandwidth), so a chip with more bandwidth per dollar than a GPU gains, which is the Ethash shape the design avoids. (3) The mix: the hour-to-hour spread is 7 to 22 percent of the median per card (the 5090 the widest, because its 64-byte loads are the expensive ones and their count per program runs 2 to 8 of 16); the programs with many 64-byte loads (mixA-3, mixA-5, mixB-2) are the slow hours on the 5090 and the fast ones nowhere. (4) The scratch: on the 5090 every RMW share costs 12 to 48 percent against the persistent control, the 32 KiB arena less than the 128 KiB one (the smaller arena, 64 MiB over 2,048 warps, sits inside the 96 MB L2); on the M5 Max the 32 KiB rows are FASTER than the control (+12 and +74 percent at 25 and 50 percent), because the arena (128 MiB over 4,096 warps) lives in the chip's caches and a scratch op is cheaper than a dataset read, so replacing dataset loads raises the rate: the scratch at these sizes is not memory work on Apple and is partly cached on NVIDIA. The chip row for these variants comes from the ca2-soundness branch; what this entry gives is the GPU cost and the share. (5) Latency-bound shares: v2 0.87 to 1.01 on the three cards, w16 0.84 to 1.03, w64 0.58 (5090) and 0.78 (9070 XT); the Mac's shares above 1 are an Apple OpenCL probe under load against a Metal rate.
Jobs: `run-readwidth-5090-20261005` and `run-readwidth-9070-20261005` (probes; the packs refused for their string seeds, fixed in a9e002c), `run-readwidth-5090-20261005c`, `run-readwidth-9070-20261005c` (benches), `run-readwidth-9070-scratch-20261005d` (the scratch packs after the `__local` fix d0018cf, AMD's compiler requires the exchange buffer at the kernel's outermost scope); read back with `node tools/jobs.mjs <id> --all`. Mac commands and logs: `docs/plans/read-width.md` section 3. The worker exes for the jobs: `proto-cuda/nvrtc/build-windows.sh` on this branch (mingw), sha256 of the CUDA one `6f46336f...defe1`.
## 5 October 2026 (night), the hot table on the M5 Max: a second table sized to GPU cache beside the 1 GiB dataset (Counter ASIC 2.0 layer 5)
Branch `ca2-cache` (on readwidth 1ea7a52), `docs/plans/hot-table.md`. Apple M5 Max, measure lock held, the Mac's load average 14 to 27 throughout (other agents' CPU work; the lock serialises builds and measurements, not every process), so the ratios inside one session are the result and the absolute rates are not quiet numbers. Packs `proto-cuda/packs-ca2-hot/hot{32,64,96}k4`, `hot64k2`, `hot64k8` from `igneum-pow export --seed igneum-genesis --day 2026-10-03 --class hot<S>k<k>` (the version 2 genesis program with k of its 16 loads redirected to an S MiB table H keyed by `seed_words("igneum-hot/" || seed bytes)`, read at `H[mulhi(src, words)]`).
**Probe** (`proto-opencl/igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S`, Apple OpenCL, wall time, best of 3, 256 dependent steps per lane, work-group 256; the ceiling row is 4,194,304 lanes):
| MiB | chase at 4,096 lanes | ns per dependent load | chase ceiling, G loads/s | indep x8 ceiling | stream |
|---|---|---|---|---|---|
| 32 | 3.51 G/s | 1,168 | 21.7 | 21.8 | 138.7 GB/s |
| 64 | 3.63 | 1,129 | 12.8 | 13.0 | 199.1 |
| 96 | 3.30 | 1,242 | 12.3 | 12.7 | 242.6 |
| 1024 | 2.22 | 1,844 | 3.50 | 3.50 | 521.5 |
**Hash rate and bit-exactness** (Metal `proto-metal/packbench --pack <dir> --batches 5 --batch-log2 24`, GPU time; Apple OpenCL `--bench-pack --pack <dir> --batches 5 --batch-log2 24`, wall; both fill H on the device from the pack's `igneum_hot_fill` and check it; fingerprint = FNV-1a 64 over 2^24 outputs at base 0):
| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table head, last line, FNV | g against v2 (Metal) | probe-predicted g | ideal g |
|---|---|---|---|---|---|---|---|---|
| igneum-genesis-mh (v2) | 27.68 | 27.61 | 25f96e7dce90bd4e | 96/96 both | none | 1 | 1 | 1 |
| hot32k4 | 33.90 | 33.92 | d2e6cf3b61d0b9fe | 96/96 both | PASS both | 1.22 | 1.27 | 1.33 |
| hot64k4 | 30.93 | 30.89 | e4c5263ac650cc0d | 96/96 both | PASS both | 1.12 | 1.22 | 1.33 |
| hot96k4 | 29.06 | 28.97 | 5d63439b6e394521 | 96/96 both | PASS both | 1.05 | 1.22 | 1.33 |
| hot64k2 | 27.67 | 27.27 | 352633bdbbb0d2b6 | 96/96 both | PASS both | 1.00 | 1.10 | 1.14 |
| hot64k8 | 47.42 | 46.73 | da54630d7dfaaf85 | 96/96 both | PASS both | 1.71 | 1.57 | 2.0 |
Hot table fill, Metal GPU time: 0.07 ms (32 MiB), 0.15 (64), 0.22 (96). Hot table FNV-1a 64 of the genesis epoch: c1767ba3ef02719f (32 MiB), 77ca4b9527104530 (64), 79bcf436c4e5bc47 (96); cache 48c4f5bf24166b2e unchanged.
**CPU verifier** (`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class <c> --warps 50`, one core, release):
| Class | hot fill, one core | items per warp | ms per warp |
|---|---|---|---|
| v2 | none | 4,096 | 0.626 |
| hot32k4 | 24.0 ms | 3,072 | 0.489 |
| hot64k4 | 46.4 ms | 3,072 | 0.504 |
| hot96k4 | 73.0 ms | 3,072 | 0.488 |
| hot64k2 | 45.5 ms | 3,584 | 0.560 |
| hot64k8 | 47.7 ms | 2,048 | 0.344 |
Reading: bit-exact across Metal, Apple OpenCL and the Rust reference on every hot pack, hot table included. On this card the 32 MiB table delivers 92% of the probe's predicted gain with the dataset streaming beside it, 64 MiB about half, 96 MiB a quarter; k = 8 at 64 MiB gives 1.71x against an ideal 2.0x. The verifier gets cheaper with k (a hot load is one table read, a dataset load is an item derivation) and pays 24 to 73 ms per epoch for the fill. Chip model with these g in the plan, section 6.4. The RTX 5090 and RX 9070 XT rows are a prepared PC job (`relay/playbooks/ca2-hot-{5090,9070}-bench.ps1`, zip `~/Desktop/igneum-ca2-hot.zip`), not run.
Crate: `cargo test --release` 52 pass (39 unit, 13 pack tests: the four pinned v2 packs byte-identical, the five hot packs pinned with their load-form count: exactly 16 - k masked dataset loads and k hot loads per hash kernel).
**Addendum, the added form** (coordinator's form of 5 October 2026: 16 + k load slots, the k hot ones drawn among them, the 16 dataset loads and the 4,096-item verifier bound unchanged; packs `hot32k4a`, `hot64k4a`, `hot96k4a`; second Mac session 21:03 to 21:19 UTC, load average 7 to 14; same harnesses and commands, branch `ca2-cache` on ca2-v3 464d6e1, the hosts rebuilt on the merged packfile.h):
| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table | g against v2 (Metal, v2 27.63 in this session) | probe-predicted g | CPU verify ms/warp (v2 0.602) | hot fill, one core |
|---|---|---|---|---|---|---|---|---|---|
| hot32k4a | 25.76 | 25.72 | 8a3414735db4523c | 96/96 both | PASS both | 0.93 | 0.96 | 0.631 | 21.7 ms |
| hot64k4a | 23.92 | 23.87 | 45668f34105f6307 | 96/96 both | PASS both | 0.87 | 0.94 | 0.609 | 43.3 ms |
| hot96k4a | 22.92 | 22.88 | af763997dfee4c82 | 96/96 both | PASS both | 0.83 | 0.93 | 0.614 | 64.9 ms |
Reading: the added form costs this card 7, 13 and 17% of its rate at 32, 64 and 96 MiB for four extra loads per iteration, more than the probe predicts as the table grows; the verifier is unchanged (4,096 items, plus 32 table reads) and pays the fill per epoch. Chip arithmetic in the plan, section 6.4. All eight packs load and self-test through the rebuilt OpenCL host (the Windows exe's host.c) on the Mac.
**Addendum, the PCs** (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7, app 0.3.9 before and after; fetch `fetch-ca2-hot-20261005` (zip sha256 bd49faa1c9d48024f49c615481faff5c68a4c09f0889dbaf009c208674d67b3f), jobs `run-ca2-hot-5090-20261005` (126 s) and `run-ca2-hot-9070-20261005` (247 s), both exit 0, the card under test switched off in the app through `api/cards` and restored; workers `igneum-worker-cuda.exe` sha256 956c4ab34f42cbcd1d2c1c6fb1a58fd9b3a8c70166df771cafcd0296ca6a27d4 and `igneum-worker-opencl.exe` sha256 32d3d34390aad70485c3524424c354223387137d383b5c5daf01f40073c12703, built from ca2-cache 196db96 on ca2-v3's merged packfile.h d2cd6e1; read back with `node tools/jobs.mjs <id> --all`):
Probe (`--memprobe --probe-mib S`, dependent 4 B chase ceiling at 4,194,304 lanes, G loads/s; ns per dependent load at 4,096 lanes in brackets):
| Card | 32 MiB | 64 | 96 | 1024 | stream at 1024 MiB |
|---|---|---|---|---|---|
| RTX 5090 (CUDA, wall) | 112.6 (320) | 112.6 (340) | 112.6 (336) | 17.6 (610) | 1,563 GB/s |
| RX 9070 XT (OpenCL, event) | 9.88 (396) | 9.47 (457) | 8.18 (454) | 2.43 (1,579) | 633 GB/s |
Rates (5 dispatches of 2^24 after a warm-up; 5090 `--bench --block-warps 1`, 9070 XT `--bench-pack --device 1` work-group 256; every row check=PASS with the Mac's fingerprint; v2 references from the readwidth entry, same night, same workers: 136.1 and 18.15 MH/s):
| Pack | 5090 MH/s | g | 9070 XT MH/s | g | ideal g |
|---|---|---|---|---|---|
| hot32k4 | 146.6 | 1.08 | 19.79 | 1.09 | 1.33 |
| hot64k4 | 140.8 | 1.03 | 18.73 | 1.03 | 1.33 |
| hot96k4 | 138.5 | 1.02 | 18.33 | 1.01 | 1.33 |
| hot64k2 | 137.5 | 1.01 | 18.17 | 1.00 | 1.14 |
| hot64k8 | 163.6 | 1.20 | 22.32 | 1.23 | 2.0 |
| hot32k4a | 118.7 | 0.87 | 15.27 | 0.84 | 1 |
| hot64k4a | 115.4 | 0.85 | 14.62 | 0.81 | 1 |
| hot96k4a | 114.4 | 0.84 | 14.56 | 0.80 | 1 |
Reading: the probe promises a full hit rate on the 5090 (every S inside the 96 MiB L2 at one ceiling, 6.4x DRAM) and the hash gets 2 to 8% at k = 4 and 20% at k = 8; the 9070 XT the same shape. The dataset's random lines evict the table from the shared cache on every card. The added form costs 13 to 20% of the rate. Recommendation in `docs/plans/hot-table.md` section 6.4: do not adopt layer 5 in either form on these measurements.
## 5 October 2026 (night), mixer x4 and the cache growth rule: the class v3 dataset construction, with the x8 candidate (Counter ASIC 2.0; branch ca2-mixer on ca2-v3 6c75dad; cryptographer's lane)
Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0. Write-up `docs/plans/mixer-x4.md`; chip model `docs/analysis/chip-model-v3.md`; code `igneum-pow` (LoadClass mixer_mult and growth, memhard::Shape, the schedule, the three emitters), packs `proto-cuda/packs-ca2-mixer/`, tests `igneum-pow/tests/mixer.rs` and `tests/packs.rs`. Commits 0fc0ad1, 66eeba3, e4c04a7, 7ce8d1e, 504cae4, fe4e193 and this entry's.
What changed. Under program class v3 (`V3_CLASS = LoadClass::MX4`) every mixer application of the item derivation is `m = 4` applications with round keys `(r m + j + 1) x 0x9E3779B9`, the 8 dependent cache reads per item unchanged; the cache doubles when the dataset doubles (`growth_doublings(d) = floor(log2(1 + d / 1460))`: 2^26 words to day 1,459, 2^27 from day 1,460, 2^28 from day 4,380). Version 2 is byte-identical: fresh exports of igneum-genesis-mh and igneum-devnet-v4-epoch0 `diff -r` IDENTICAL against the checked-in packs, and the crate tests regenerate every pinned file. A v3 program of a seed is the v2 program of that seed instruction for instruction (v2 loads take no width roll); only the dataset words and the hashes change.
Bit-exactness, `with-lock.sh run`, 22:05 and 21:45 UTC: the two pinned v3 packs (mx4-genesis, mx4-devnet-epoch0: dataset words 0..15 `61ff2180 0d4c7e6c ...` and `afe80d67 b9fbd029 ...`, word MASK `5020180e` and `e6a99c7a`, unit at base 0 lane 0 `63acd2d273f475ba` and `212c6442b51e87ae`) and the two x8 candidate packs on Metal (`packbench`, built from this branch) and Apple OpenCL (`igneum-bench-cl --bench-pack`): 3/3 standalone and 3/3 in batch, 96 of 96 lanes, cache FNV-1a 64 unchanged from v2 (48c4f5bf24166b2e, 448274a57f508cbc), dataset head, word MASK and 64 samples PASS, one 2^24 fingerprint per pack across both harnesses (mx4 6f48d5a2aa0dbe5f and 73caaebb28e808fe; mx8 7c28cfb06c5c65a9 and bbb183f72692f840); hash rate the v2 rate (27.5 to 27.7 MH/s GPU time, the hash kernel is unchanged). Fuzz: 200 class v3 programs (4 units each across the 32-bit range, one in the top 256 nonces) interpreted twice on the CPU, 800 of 800; the same 200 packs on Metal 200 of 200 (`--batch-log2 9 --batch-base 4294967040`, the wrapping unit inside the window), every tenth on Apple OpenCL 20 of 20; x8: 50 of 50 on Metal, 5 of 5 on OpenCL. Stats (8,192 outputs per seed, two seeds): v3 avalanche 49.97 to 49.99 percent, worst bit z 1.92 to 3.09, 0 duplicates (v2 beside it 49.87 to 49.98, z 2.25 to 2.30). Edges: items 0, 1, 2^28 - 1, 2^32 - 1 by hand at m = 1, 2, 4, 8; words 0, 15, 16, 17, MASK - 1, MASK through the fetch path. Determinism: two epochs, every vector and file equal and equal to the pinned pack. The scratch soundness tests of ca2-soundness (cherry-pick 0d8f745) 7 of 7 on this tree. Crate: 44 lib + 12 packs + 4 mixer + 7 scratch tests pass. A first Metal fuzz run reported 200 of 200 FAIL on an empty RESULT line (a packbench built before the `--batch-base` cherry-pick); it was read as a failure, the harness rebuilt, the run repeated.
Timings, `with-lock.sh measure`, one session 21:40:12 to 21:40:23 UTC, one core, two rounds; the box carried a load average of 5.6 (one minute) and 26 (fifteen minutes) from unlocked processes, so the absolute figures are about 2.2x the quiet readwidth night's 0.604 ms v2 row and the ratios are the measurement:
| Construction | Verifier ms per 32-lane unit, avg of 50 (two rounds) | Worst cold unit | Against v2 | 256 MiB fill, one core | Metal 1 GiB build, GPU ms |
|---|---|---|---|---|---|
| v2 | 1.361 / 1.310 | 1.579 | 1 | 172 to 173 ms | 29.7 (first touch) / 21.0 |
| x4 (class v3) | 1.956 / 1.923 | 2.043 | 1.45x | 172 to 175 ms | 20.9 / 21.0 |
| x8 (candidate) | 2.785 / 2.790 | 2.942 | 2.09x | 172 ms | 21.9 / 21.9 |
Reading: the mixer multiplies the verifier's ALU part only (the 8 dependent misses per item are unchanged), hence 1.45x and 2.1x and not 4x and 8x; the Mac's GPU build is latency-bound and does not move with the mixer, so the "under 1 s on every discrete card" half of the x8 rule is the PC job (five packs, `relay/playbooks/mixer-x4-pc1-bench.ps1`, waiting for the go). Verification throughput (C19): a quiet 2026 core serves about 1,100 shares per second at x4 and 800 at x8 (1,660 at v2, re-cutting spec 09's 2,270), a 22,000-member pool at one share per 10 s needs 2 cores at x4 and 3 at x8, IBD over 108,000 headers is 1.6 min at x4 and 2.3 at x8 on that core; the 10 ms gate keeps 8.0 ms (x4) and 7.1 ms (x8) of margin on the loaded core, 6 to 7 ms on a 2019-class laptop core (approximate, unmeasured, O-1.14).
Chip model (`docs/analysis/chip-model-v3.md`): the on-die-cache recompute chip at 50 T op/s against the 5090's measured 136.1 MH/s: v2 334 MH/s, 2.45x bare, 7.4x with the 3x fixed-function factor; x4 83.5 MH/s, 0.61x bare, 1.84x with the factor, 1.53x with the 128 mm^2 N5 mirror deducted at equal silicon; x8 41.7 MH/s, 0.31x, 0.92x, 0.76x. The claim at x4 is "under 2x" with the margin thin on the equal-budget convention (a 3.3x factor or a 10 percent larger budget reads 2.0x); the hot table in the added form would have raised it to 2.1x to 2.2x at the 5090's g (kept as measured, not adopted). Nothing here is a measurement of a chip.
**Addendum, 22:15 UTC: the verifier regression, the PC 1 build rows, and x8 into v3.** The era agent measured the same v2 input with readwidth's binary (0.604 ms) and ca2-v3 HEAD's (1.33) in one minute; bisected under the measure lock to this branch's 0fc0ad1 (seam 6c75dad 0.610, 0fc0ad1 1.332; the "loaded box" reading above was wrong by that factor, the load was real but the 2x was the code). Cause: the item loop (`derive_items`) inlined into `MemhardCpu::fetch`; the mask hoisted, the mask constant, and the constant-mask loop inlined all stayed at 1.33, the same loop `#[inline(never)]` read 0.60 to 0.62. Fix: `derive_items_mask`, out of line, one instance per cache size with the line mask a constant. Measured the era agent's way (readwidth's binary beside the fixed one, same input, same minute, 22:07 UTC): v2 0.607 / 0.610 against 0.609 / 0.611; on the fixed binary x4 1.238 / 1.237 (2.0x), x8 2.077 / 2.058 (3.4x), worst cold 2.15 ms; the increments (+0.63, +1.46 ms per unit) equal the slow binary's. Lesson, the class: an inlined item loop costs 2.2x and nothing in the suite sees it; a verifier benchmark with a pinned bound in the crate's CI is filed for the next cut, and until then every change to the item loop is measured against the previous binary on the same input in the same minute. PC 1 (job run-mixer-x4-pc1-20261005, 22:00 to 22:04 UTC, the worker's `cache ... dataset ... ms` wall line): RTX 5090 dataset 23 to 25 ms at v2, x4 and x8; RX 9070 XT (gfx1201) 72 to 77 ms at all three; every fingerprint equal to the Mac's; rates the v2 rate (136.5 to 137.4 and 18.0 to 18.2 MH/s). Decision under the delegated rule (coordinator, 22:05 UTC): x8 enters class v3 (`V3_CLASS = MX8`); pinned packs mx8-genesis (7c28cfb06c5c65a9) and mx8-devnet-epoch0 through the chain path with the era inside (90f794dd556f7a3b, Metal and Apple OpenCL, 22:12 UTC); the x4 packs kept as the candidate's record. Chip headline at x8: 41.7 MH/s, 0.31x bare, 0.92x with the 3x factor, 0.76x at equal silicon (`docs/analysis/chip-model-v3.md`).
## 5 October 2026 (evening), EVM transaction relay: three nodes in a chain, every transaction sent to one end included by the other two miners (execution and networking engineer)
Until this change the node did not relay EVM transactions to its peers, so a transaction sent to one node was only ever included by that node's own templates (this file, "5 October 2026 (afternoon), live devnet: real transactions": 3,794 transfers, all in the Mac's blocks; execution-layer ledger item 9). Fork branch `tx-gossip` (worktree `vendor/igneum-node-txgossip`, from release-0.3.6 a24ab01a, commit e242acd0), main repo branch `tx-gossip`. Design in `docs/design/execution-layer.md` 1.4 "Relay"; the hand-out cooldown of its 10.2 table is gone with it (row "Mempool hold").
@ -1586,3 +1728,236 @@ The 3-node run (`tools/txgen/relay-net.mjs`, new; this Mac, load 7 to 8 at the e
Reading. Every transaction given to A was mined by B or C within 6 s, two thirds of them within 3 s, with no skipped copy: the hold on block-added kept B's and C's parallel blocks from carrying the same transfer. The afternoon run on the live devnet, through one node with the cooldown, had p50 40.7 s and p90 110.8 s with 50-s quiet stretches; here the 1.5 s p50 is one fast-time block plus the relay and the executor's lag. The pool depth matching on all three nodes at every sample is the convergence. Not measured here: a transaction flood above the per-peer rate (the bucket is unit-tested only), a 13 peer in the fleet (the digest check and the version gate are the evidence), and the hold's 30-s expiry on a block that never reaches the chain (not seen in 149 chain blocks).
Commands: `IGNEUMD=vendor/igneum-node/target-txgossip/release/igneumd IGNEUM_MINER=vendor/igneum-node/target-txgossip/release/igneum-miner tools/lock/with-lock.sh run node tools/txgen/relay-net.mjs --rate 2 --duration 120 --wallets 16 --fund 2`; the Mac binaries from the fork worktree with `CARGO_TARGET_DIR=vendor/igneum-node/target-txgossip cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` under the build lock (an APFS clone of `target-036`, 2 min 15 s to clone, 5 min 06 s to build); the suites with `node tools/build-job.mjs run --target 1ccfe586 --node vendor/igneum-node-txgossip --targets linux --node-tests "igneum-exec kaspa-p2p-flows" --no-app`.
## 5 October 2026 (night), the SP1 CPU prover on PC 1 beside the miners, and the backend survey: no zkVM proves on AMD (amd-prove agent)
Josh, 22:50 BST: "test proving on the amd card?" and "can we test proving on mac?". The analysis with the backend table and the tier consequences: `docs/analysis/amd-proving.md`. The survey (SP1 v6.8.1 and `dev` 318dd530 of 28 Sep 2026, RISC Zero, Jolt, OpenVM, ICICLE, sppark; every claim cites a file or page there): on 5 October 2026 no zkVM proves on an AMD GPU; Apple silicon has RISC Zero's shipped Metal prover and ICICLE's Metal backend; SP1, the prover here, is CPU-only off NVIDIA.
Machine: PC 1 (machine ae432dc7), Windows 11, WSL2 Ubuntu 24.04 as root, 16 cores and 46,994 MB visible to the VM, the Igneum Miner app 0.3.9 mining on the RTX 5090 and the RX 9070 XT throughout (the 5090 at 89% mean utilisation, 59 to 70% minimum, from a 1-s `nvidia-smi` sampler under the run: the job never touched a card). Signed `run` job `cpu-prove-pc1-small2` (`tools/amd-prove/pc1-cpu-prove.ps1`), 20:49:00Z to 20:59:49Z: the hosted package `igneum-prove-wsl2-pv1b.zip` (sha256 df50dee5...) built WITHOUT the `cuda` feature (6 s warm; the first job `cpu-prove-pc1-small` built it cold in 126 s), `--mode id` the pinned pair (shard `0x2b1a81cb...`, aggregator `0x474678f3...`, pinned 2026-10-05T16:20:38Z), `SP1_PROVER=cpu`, `--mode shard --shard 0` under `/usr/bin/time -v`. Log: `node tools/jobs.mjs cpu-prove-pc1-small2 --all`.
| Fixture | SP1 cycles | Setup s | Core prove s (bytes, verify s) | Compressed prove s (bytes, verify s) | Wall s | Peak RSS | CPU |
|---|---|---|---|---|---|---|---|
| block-56-transfers-3shards shard 0 (200 pgas, one transfer) | 315,479 | 22.75 (client 19.46, shard keys 1.85, aggregator keys 1.44) | 82.5 (7,310,257, 0.210) VERIFIED | 199.2 (1,272,897, 0.035) VERIFIED | 312.1 | 29,503,652 kB (29.5 GB) | 978% (9.8 of 16 cores), user 2,516 s, system 537 s, load max 11.3 |
| block-78-increment (2 transactions, 1 executed 1 skipped) | 631,127 | 21.75 | 87.0 (7,317,857, 0.209) VERIFIED | 202.3 (1,272,897, 0.034) VERIFIED | 322.3 | 30,517,916 kB (30.5 GB) | 979%, user 2,616 s, system 541 s, load max 13.1 |
| block-338-shard1 (one shard at `S_p`, 60.8 M cycles) | | not run: the PC 1 scheduler kept the machine for the Counter ASIC 2.0 gates (21:05Z). Approximate extrapolation: about 29 SP1 shards of 2^21 cycles at about 80 s each, 40 min of core proof, then hours of recursion; floor from the 5090's ratios (6x core, 4x compressed, block-78 to `S_p`): 9 min core, 13 min compressed | | | | | |
For comparison (this log): the Apple M5 Max CPU on 4 October, loaded, block-56 shard 0: core 83.1 s, compressed 272.3 s; on 3 October the v0 guest on block-78: core 22.0 s, compressed 55.7 s. The RTX 5090: block-78 core 1.4 s, compressed 2.7 s (4 October, mining paused); a full shard at `S_p` compressed 10.9 s alone and 33.0 s beside the miner; an empty live shard 7.0 to 7.7 s beside the miner (5 October). No fresh Mac run tonight: the measure lock was held from 20:31Z (a read-width `packbench`, three builds, a 1,500-s proving-v1 network under `run`) and did not free inside the 10-minute window set for it.
Reading, and the consequences (CLAUDE.md, every number). Doubling the cycles added 4.5 s to the core proof and 3.1 s to the compressed proof: about 280 s of a CPU proof is fixed cost in the compressed-proof recursion, so no shard size brings a CPU proof under the launch deadline (20 to 60 s behind the tip) or near the 10-s assignment window; it fits only the v1 unproven deadline (600 s), which pays a CPU prover only when no card has proven the shard in 10 minutes. The 29.5 to 30.5 GB peak RSS means the CPU prover needs 32 GB free: a 64 GB Windows PC (WSL2 takes half the host's RAM by default), a 32 GB Linux machine, a 64 GB Mac; a 16 GB machine cannot run it at all. Per tier: an AMD-only home miner (8, 12 or 16 GB, Windows or Linux) mines and does not prove, and loses the 20% proving-pool share; Apple silicon the same (the M5 Max mines at 26.7 MH/s, this log, 4 October); a mixed rig proves on its NVIDIA cards and the rig installer's `prover_decision` already skips every non-NVIDIA card (`packaging/linux/bin/igneum-rig-lib.sh`, branch `rig-install`), now a stated requirement; the app's `provedefault.rs` already keeps proving off on Apple silicon and off without an NVIDIA card. Decision asked of nobody: no CPU tier (the analysis, section 4a); the public line for the site, litepaper and Proving tile is in section 4c ("Proving needs an NVIDIA card with 16 GB or more today ... AMD and Apple cards mine. A prover for them lands when a zkVM ships one"). The first job proved nothing because an apostrophe inside a single-quoted awk program ended the quote and bash refused the loop while the job reported exit 0; the class fix is `tools/amd-prove/check-job-bash.sh` (`bash -n` on the embedded bash body before publishing) and the same `bash -n` inside the job before the run, both shown to refuse the bad body and pass the fixed one.
## Counter ASIC 2.0, the numbers
5 October 2026 (night). The chip-resistance layers measured on the three cards we own (Apple M5 Max, RTX 5090 on PC 1 and PC 2, RX 9070 XT on PC 1's eGPU), the decisions taken under Josh's delegated rules for the devnet, and the chip model before and after. Every number is from an entry above or from the plan documents named; approximate is marked. Levels: `docs/plans/counter-asic-2-public.md`.
**Program class v3 (the devnet, activation by height switch `program_class_v3_activation_daa`)** = class v2's 128 x 4-byte loads, the era draw of the table layout and the working-set windows (layers 4 and 8), the cache growth rule (layer 6, option C: the cache doubles when the dataset doubles), the mixer at x8 (M16's multiplier), reserve family R1 (integer matrix, switched off) and the epoch length as a signalled reserve parameter (layer 9, 3,600 DAA s until a 90% signal). Not adopted on the measurements: wider reads (layer 1), the per-load width mix (layer 2), the per-warp write scratch (layer 3), the hot table (layer 5).
| Card | v2 MH/s | v3 MH/s, six eras (spread) | Bytes per hash | Latency-bound share | Daily 1 GiB build, v2 / v3 |
|---|---|---|---|---|---|
| Apple M5 Max, Metal | 27.68 | 27.85 to 27.98 (0.5%) | 512 | 1.06 | 21 / 21 ms |
| RTX 5090, CUDA | 137.2 | 135.90 to 137.70 (1.3%) | 512 | 1.01 | 25 / 23 ms |
| RX 9070 XT, OpenCL | 18.09 | 18.59 to 19.18 (3.1%) | 512 | 0.95 | 74 / 75 ms |
CPU verifier, one M5 Max core at load average 5.5 (the fixed crate, ca2-mixer 1ab8b21): v2 0.61 ms per warp, v3 (x8) 2.08 ms (3.4x), worst cold 2.15; the 10 ms gate holds 4.8x (4.6x on the worst cold unit). Bit-exact: every v3 pack's fingerprint equal on Metal, Apple OpenCL, CUDA and AMD OpenCL.
| Layer | Measured | Decision | The number |
|---|---|---|---|
| 1 wider reads | w16 139.8 / 17.90 / 28.26 MH/s (5090 / 9070 XT / M5 Max) against v2 136.1 / 18.15 / 27.74; w64 71.9 on the 5090 (share 0.58, 37% of its stream) | out: keep 4 B | the 9070 XT does 2.4 G dependent reads/s at every width; wider reads make the 5090 bandwidth-bound |
| 2 width mix per load | spread over six programs 18.8 / 7.4 / 11.3% and 22.3 / 5.5 / 8.1% | out | the 5% rule |
| 3 write scratch | GPU cost 12 to 48% at 32 and 128 KB per warp; the on-die-cache chip 2.4x at every share | out (the construct is sound; its tests stay) | the verifier resets the scratch per unit, so a chip keeps it in 80 to 320 B per lane |
| 4 and 8 era layout and windows | six-era spread 1.3 / 3.2 / 0.8% | in | under the 5% rule; the SRAM mirror a chip needs is the whole dataset every hour |
| 5 hot table | added form g 0.87 / 0.85 / 0.84 (5090), 0.84 / 0.81 / 0.80 (9070 XT) at 32 / 64 / 96 MiB | out (a 3.0 option) | no card keeps 32 MiB resident while the dataset streams; the replaced form helps the chip |
| 6 cache schedule | the 256 MiB mirror is 128 mm^2 and $46 at N5 by shipped cache-die density, approximate | option C, in | the cache's job is to stay above GPU L2 (96 MB on the 5090, 128 MB on GB202) |
| 7 integer matrix | dp4a 1.17x a step on the 5090, 1.06x on the 9070 XT, 1.6x emulated on Apple; mm8 native on all three as a tile | reserved R1, off | unlock at era 4 or 90% signal |
| M16 mixer | x4: verifier 1.24 ms, chip 1.84x with the allowance; x8: 2.08 ms, 0.92x; the daily build unmoved on every card | x8 in | the only lever that moves the named chip |
| 9 epoch length | compile-ahead 0.5 s (M5 Max, race off), 1.0 s (5090), 38 s with the race; FPGA compiles 42 to 160 min (PRflow, FPT 2019) | reserved, 600 s to 2 h by signal | at 600 s a per-program bitstream mines 0% of each epoch |
**The chip model, before and after** (`docs/analysis/chip-model-v3.md`, `docs/analysis/sram-mirror.md`): the strongest chip we can name holds the whole 256 MiB cache on-die (about 128 mm^2 and $46 of silicon at N5, approximate) and computes dataset items on the fly at 50 T integer op/s. Against the RTX 5090's measured 136.1 MH/s: class v2 333 MH/s, 2.4x; class v3 41.7 MH/s, 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon. The claim is "under 2x"; the margin is thin on the allowance (3.3x reads 1.0x) and 9% on the budget. Next levers, named: the mixer at x16 (the verifier at about 4 ms per warp; a 2019-class core unmeasured), a hot table small enough to stay resident beside the streaming dataset.
**The user tiers.** AMD RDNA 4 sits at about a seventh of a 5090 on this hash (its dependent-read rate: 2.4 G against 17.5 G per second), 2.2x worse per pound at list prices and 4.9x worse per watt (approximate); the card's memory system, not a tuning gap. The integrated tier on the CUDA and OpenCL one-click workers mines v3 with a restart per epoch until per-day dataset reuse lands (0.3.12). Card lifetime under the step schedule: a 4 GB card to year 4, 8 GB to year 12, 12 GB to year 28 with the cache freed after the daily build.
**The bounty.** A bounty for any chip design beating a GPU by more than 2x on the published model, with a leaderboard by card model, follows the external review (spec O-1.17, January 2027); it is named publicly only once escrowed (`docs/plans/funding.md`, rule 3), which it is not yet.
## 5 October 2026 (evening), proving v1: segment records, the chain rule, the unproven rule; what was measured tonight (proving engineer)
Branches `proving-v1` (main repository, worktree `igneum-wt-proving-v1`; fork `vendor/igneum-node-pv1` from a24ab01a). Rules: spec 7.8; plan `docs/plans/proving-v1.md`. Every row names its command. The live devnet was in a degraded state the whole evening: from 18:35Z the RTX 5090 workers on PC 1 and PC 2 exited at start on a pack seed mismatch (`the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT`, restart 60+ on PC 2 by 19:05Z, another agent's branch `pack-loop`), the Mac app node was down from 17:45Z, so PC 2 mined 3.4 MH/s from its iGPU and PC 2's prover was the only prover; the coordinator held every PC 2 measurement at 19:00Z until the fleet mines again.
### Step 1, the prover default and its cost
| What | Measured |
|---|---|
| The default rule (`app/igneum-app/src/provedefault.rs`) | `cargo test --release -p igneum-app provedefault` on this Mac (the app crate, build lock, 19:05Z): 5 passed (a 5090 with WSL2 on Windows is on; Windows without WSL2 off with the Set up hint; Linux needs no WSL2 and the 12 GB gate holds, a 10 GB 3080 and a 16 GB AMD card stay off; Apple silicon off; the biggest qualifying card is named) |
| Mining alone against mining with the prover, first try (PC 2 job `prover-cost-pc2-pv1`, `tools/proving-v1/pc2-prover-cost.ps1`, published 18:40:48Z, ran 18:41:13Z) | VOID: the job waited 20 min for the 5090 worker to hash and it never did (the pack fault above); "mining alone" was 0 MH/s |
| Mining alone against mining with the prover, the re-run after the coordinator's go (job `prover-cost-pc2-pv1b`, ran 19:20:36Z to 19:51:17Z; the 5090 worker restored at 19:16Z and hashing throughout; prover OFF by `POST /api/prove {"on":false}` 19:40:39Z, back ON 19:46:09Z, left on). The job's own `/api/state` samples stayed empty on PC 2 (`Invoke-RestMethod` returns an object PowerShell 5.1 cannot walk, `cards=0`, the fix is for the next run), so the hash rate is read from the miner's own STATUS lines (`miner-nvidia-1ccfe586-1` uploads, `now=... MH/s wall`, one every 30 s, the intake table `miner_logs`) | prover OFF, 19:41:09 to 19:46:09Z: n 10, mean 124.72 MH/s, p50 124.81, min 124.10, max 125.38. Prover ON, 19:46:39 to 19:51:13Z: n 9, mean 119.74, p50 118.87, min 118.08, max 123.42. The 15 min before the job with the prover on (19:25 to 19:40Z): n 30, mean 119.88, p50 118.79. So the prover costs the 5090 5.0 MH/s, 4.0% of its hash rate, while it proves the devnet's empty shards one after another (1.4 a minute here: the node's paidShards 559 -> 566 over the 5-min phase). A full shard at `S_p` keeps the card busier (the 4 October run proved one in 10.9 s); the cost at that load is the chain job's row |
| GPU memory during proving, first try (phase B of the first job: the prover on for 5 min, 298 one-second `nvidia-smi --query-gpu=memory.used` samples, the 5090 worker dead so the card held nothing else) | memory.used min 1,654 MiB, max 13,816 MiB, utilisation mean 2.7%, power max 190.6 W: the prover alone on empty shards |
| GPU memory with the miner AND the prover on the card (the re-run's phase B, 298 one-second samples, 19:46 to 19:51Z) | memory.used min 3,396 MiB (the miner's dataset and program resident), max 15,590 MiB, utilisation mean 92.9%, power max 328.6 W. So the prover's own peak is about 12.2 GB on an empty shard (15,590 minus the miner's 3,396), and the two together need 15.6 GB: a 16 GB card (5080, 9070 XT class, if it had a CUDA path) sits 0.4 GB under tonight's peak with no room for a full shard, a 24 GB 4090 has 8.4 GB of headroom, a 12 GB card cannot mine and prove at once on this build. The full-shard peak is the chain job's row |
| Shards per minute with the mining worker dead | the node's `paidShards` 510 -> 518 over the 5-min phase: 1.6 shards a minute from one 5090 through the app's loop (export, cut, prove, sign, submit) |
| Host RAM (Windows `Win32_OperatingSystem` and the `vmmem` working set, sampled every 15 s) | host used 25,550 MB of 63,132 MB at the end; the WSL2 VM's working set 7,915 MB (2,334 MB used of 30,914 MB inside the distribution) |
| The SP1 GPU server's compiled targets (`cuobjdump --list-elf /root/.sp1/bin/sp1-gpu-server` inside PC 2's Ubuntu-24.04, CUDA 12.8, driver 610.47) | `sp1-gpu-server` 6.8.1 (251,306,680 bytes, sha256 c2642ad1c42e85d8525159cf0c7cd5200d8766c9be1283f452a1f9bf9fea725c, the asset `sp1_gpu_server_v6.8.1_x86_64.tar.gz` the SDK downloads, `sp1-cuda-6.8.1/src/server.rs`): one ELF each for sm_80, sm_86, sm_89, sm_90, sm_100 and sm_120; `strings` finds compute_120 PTX as well. So sm_89 (Ada: RTX 4090, 4080) is compiled in natively, no JIT; so are Ampere (3090, 3060), Hopper, Blackwell datacentre (sm_100) and consumer (sm_120, the 5090). Nothing for AMD (no HIP path in SP1) |
### Step 2, aggregated chains
| What | Measured |
|---|---|
| The new host (`--mode chain`, `aggregate`, `verify-segment`) against every fixture natively | `igneum-prove-host <f> --mode native` on the Mac for the 12 fixtures of `proving/fixtures/` (9 block, 3 fee-switch), host built from this branch 19:06Z: every one MATCHES (the package gate's native half); `--mode id`: shard `0x2b1a81cb...`, aggregator `0x474678f3...`, the 0.3.9 pin, unchanged |
| Eight consecutive live fixtures | `igneum_exportSegments 0x0..0x13cb4` on node 1's exec RPC (127.0.0.1:26790, read-only, 20:06 BST, tip 81,076): 71,042,616 bytes, 81,077 segments, 28 accounts, 0.5 s; `igneum-prove-export export.json <n> block-<n>.json` for 81046..81053: replayed 81,077 segments from genesis in 1.8 s each, every state root equal to the node's; one shard a block, 0 pgas (no transactions on the devnet tonight), `proving/fixtures/chain/` |
| Chain of 2 on the Mac CPU (the known-finished case of `--mode chain` before the GPU; M5 Max under the live nodes, the harness and two builds) | `SP1_PROVER=cpu igneum-prove-host --mode chain --chain block-81046.json,block-81047.json --out results.json` under the run lock, 19:07:48Z to 19:11:28Z: setup 12.2 s; block 81046: shard 0 compressed 55.4 s (1,272,897 bytes, verify 0.036 s), aggregate 52.0 s (1,272,909 bytes, verify 0.031 s), chain_len 1, agg_vk zero; block 81047: shard 41.3 s, aggregate WITH the previous block proof 59.1 s, chain_len 2, agg_vk = the pinned aggregator id; end to end 207.9 s; final proof 1,272,909 bytes, statement 0x232276f4... The recursion over the previous proof cost 7 s more than the first aggregation on this CPU |
| `--mode verify-segment` on that proof (the node's path: SP1 light verifier, pinned aggregator key) | VERIFIED in 0.032 s (0.27 s wall, three runs: 0.033, 0.032, 0.032); known-failed: a wrong statement NOT VERIFIED (0.032 s); the shard verifier (`--mode verify`) on the segment proof NOT VERIFIED, "program id 0x474678f3... IS NOT OURS 0x2b1a81cb..." |
| Chain of 8 on the RTX 5090 (N = 2, 4, 8), job `chain-pc2-pv1b` (`tools/proving-v1/pc2-chain.ps1`; the package `igneum-prove-wsl2-pv1.zip` eb6dccf8..., 1.5 MB, fetched by `fetch-prove-pv1` 19:51Z; the first try `chain-pc2-pv1` died in its own export step, fixed) | Ran 19:58:37Z: the export from PC 2's node (72,901,414 bytes, 1.4 s), the host built in WSL2 against the live build's warm target dir in 6 s and installed to `/opt/igneum-pv1` (the live `/opt/igneum` host untouched, sha 29cc4768...), `--mode id` the pinned pair; eight consecutive fixtures 83346..83353 cut, every one MATCHES natively. The chain on the GPU (SP1_PROVER=cuda, the miner mining on the same card at 119 MH/s): setup 12.7 s; block 83346: shard 7.4 s, aggregate 7.6 s (chain_len 1), 15.1 s; block 83347: shard 7.2 s, aggregate WITH the previous proof 9.5 s (chain_len 2, agg_vk the pinned aggregator id), 16.8 s, cumulative 31.8 s over 2 blocks; block 83348: shard 7.0 s, then at 20:01:09Z the app quit and aborted the job ("quit: stopping the miners, then the node", then "job chain-pc2-pv1b: aborted (the app is quitting)"; NOT an update: nothing of 0.3.10 was published; the log gives the quit no source; 20 s earlier the efficiency sweep's administrator prompt had been cancelled at the keyboard, and 13 s earlier the live prover had failed with "CudaClientError: Connect(PermissionDenied)", the root-owned socket my job had left, below). So N = 2 measured: 31.8 s of GPU time for two empty blocks, the chained aggregation 1.9 s dearer than the first; N = 4 and 8 are the re-run `chain-pc2-pv1c` after the restart. An empty shard's compressed proof on the 5090 is 7.0 to 7.4 s (the 200-pgas shard of 4 October took 2.7 s with the card to itself; tonight the miner held it at 92% utilisation) |
| The chain of 8, the third run `chain-pc2-pv1c` (20:05:21Z to 20:08:33Z, after the app restart; blocks 83616..83623 from PC 2's node at tip 83646, the same script; results `tools/proving-v1/chain-pc2-2026-10-05.json`) | Build 5 s (warm), eight fixtures cut and MATCHING natively, setup 15.7 s, then on the GPU with the miner mining on the same card: shard proofs 7.3 to 7.7 s each (8 x, 59.5 s), aggregations 7.9 s for the first block and 9.6 to 9.7 s for every chained one (75.5 s), every proof VERIFIED, end to end 135.6 s for 8 blocks (17.0 s a block from the second on). Cumulative: N = 2 at 32.6 s, N = 4 at 66.8 s, N = 8 at 135.6 s. The final proof is 1,272,909 bytes whatever N (chain_len 8, agg_vk the pinned aggregator id), the record 586 bytes; `--mode verify-segment` on it: VERIFIED in 0.039, 0.037, 0.040 s after a 0.26-s light-verifier setup, the same three runs each time. GPU memory over the chain (152 one-second samples): max 16,751 MiB with the miner's 3.4 GB resident, so the chained aggregation holds about 13.4 GB, 1.2 GB over the shard-only peak; WSL used 2,456 MB |
Reading the chain numbers. Aggregation is a fixed cost per block (9.7 s here), not per segment: the recursion verifies one more proof whatever `chain_len`, so the record for N blocks costs N aggregations and the verifier one. Against 4 October with the miner stopped (aggregate 2.2 to 2.5 s, a 200-pgas shard 2.7 s), tonight's 9.7 s and 7.3 s say the miner's 92% utilisation slows the prover about 3 to 4x while the prover slows the miner 4%: the card is shared, and the lottery wins the arbitration. A machine that mines and proves at once delivers one empty block's proof and aggregation in 17 s; one that only proves, about 5 s (approximate, from the 4 October stages).
### Step 3, coverage
| What | Measured |
|---|---|
| A 3-minute window at 18:57Z on node 1 (`node tools/proving-v1/coverage.mjs --minutes 3`, chain blocks 80754..80839, 86 blocks) | 4 blocks with a paid shard (4.7%), 4 fully proven, 4 of 86 shards; on-chain latency (carrier timestamp minus block timestamp) n 4: min 36 s, p50 39 s, max 44 s; 0 content blocks. One prover (PC 2), the Mac verifier node down, PC 2 producing few blocks (3.4 MH/s): the degraded state above, not the fleet's number |
| A 30-minute window, 19:13 to 19:43Z, the degraded fleet (PC 2 the only prover, its 5090 worker restored at 19:16Z, the Mac app node down by decision: the Mac app is attached to node 1) | `node tools/proving-v1/coverage.mjs --minutes 30 --watch` on node 1: chain blocks 81236..82668, 1,433 blocks; 38 with a paid shard (2.7%), all 38 fully proven (one shard a block, 0 content blocks); on-chain latency n 38: min 36, p50 44, p90 52, p99 62, max 65 s. The live page's 10-minute proving object read 0 shards and 0 provers at 19:42Z (it counts what its own node verified; that node is the Mac app node, down), so the chain's own count is the number |
| A 30-minute window with the fleet mining (PC 2 at 119 MH/s from 19:16Z, PC 1 at 128.8 from 19:18Z; PC 2 still the only prover, its prover OFF for the 5 min of the cost job's phase A inside this window; the Mac app node down by decision) | `coverage.mjs --minutes 30 --watch`, 19:21 to 19:51Z on node 1: chain blocks 81644..83069, 1,426 blocks; 34 with a paid shard (2.4%), all fully proven (one shard a block, no content); on-chain latency n 34: min 38, p50 44, p90 51, p99 52, max 53 s. One 5090 through the app's loop as it is covers 2.4 to 2.7% of the blocks; the latency from block to carried record is 44 s at the median, under the litepaper's minute, and would be the same for every block if the fleet were 40 cards (the table below) |
### Step 3, the fleet size (arithmetic from measured inputs; every input names its entry)
Inputs, all RTX 5090 (PC 2), SP1 6.8.1 cuda: a full shard at the provisional `S_p` (6.75 M pgas) compressed in 10.9 s and the four shards of a near-`B_p` block in 10.2 to 10.7 s each (bench-log 4 October 2026, "shard proving on the RTX 5090", runs run-20261004-173115 and run-20261004-r3-shards); one aggregation 2.2 s (two shards) to 2.5 s (four shards), the same entry; tonight's chain of 2 on the Mac CPU shows the recursion over the previous block proof costs the same order as a first aggregation (52.0 s against 59.1 s), so the GPU figure for a chained aggregation is taken as 2.5 s, approximate, until the held PC 2 chain job measures it; the app's live loop tonight: 1.6 shards a minute per card on empty shards (export, cut, key setup, prove, sign, submit: about 37 s a shard, of which the proof is a few seconds), bench-log step 1 above. A 5090 proves one thing at a time.
| Block content at 1 block/s | Shard proofs a second (fleet) | Card-seconds a second for shards | Aggregations a second | Card-seconds a second for aggregation | 5090-class cards for 100% | Rule |
|---|---|---|---|---|---|---|
| empty blocks (tonight's devnet), the app's loop as it is, the card also mining | 1 | 37 | 1 | 9.7 (measured, `chain-pc2-pv1c`) | 47 | one shard per block, the loop's 37 s each plus a chained aggregation |
| empty blocks, the chain mode's shape (one key setup per process, proofs back to back), the card also mining | 1 | 7.4 (measured) | 1 | 9.7 (measured) | 18 | 17.1 card-seconds a block, `chain-pc2-pv1c` |
| empty blocks, cards that only prove | 1 | 2.7 (4 October, a 200-pgas shard) | 1 | 2.5 (4 October) | 6 (approximate) | the miner's 92% utilisation costs the prover 3 to 4x |
| one full shard a block (`S_p`, 6.75 M pgas), cards that only prove | 1 | 10.9 | 1 | 2.5 | 14 | 4 October's stages |
| one full shard a block, the card also mining | 1 | about 35 (approximate: 10.9 x 3.2, tonight's ratio) | 1 | 9.7 | about 45 (approximate) | the full-shard proof with the miner on the card is not measured |
| blocks at `B_p` (four full shards), cards that only prove | 4 | 42.5 | 1 | 2.5 | 45 | 4 x 10.6 + 2.5 |
| at the adopted v1 budgets (`B_p` 120,000 pgas, `S_p` 30,000, from DAA 210,000 on the devnet): a v1 shard of transfers ran at 213 to 236 cycles per pgas (bench-log 5 October, "the prover carries both fee tables"), 7 M cycles a shard against 60 M for the prototype shard | 4 | under 42.5 (the 5090 time for a 7 M-cycle shard is not measured; scaling 10.9 s by cycles gives about 1.3 s, approximate) | 1 | 2.5 to 9.7 | 8 to 15 (approximate) | measure before the switch lands |
Reading. The card count is the sum of card-seconds of work per block-second, rounded up, with no slack for the exclusive window, the relay or a card's idle gaps; the devnet's own numbers tonight (one card, 1.4 to 1.6 shards a minute, 2.4 to 4.7% of blocks) are the first row. Two levers, both measured tonight: the loop (a shard's carriage through export, cut and a 12-s key setup is 25 s on top of a 7-s proof; the host's `--mode aggregate` and `--mode chain` hold one key setup per process and the prover loop should do the same, the 0.3.11 item in the plan) and the card's other job (a mining card proves 3 to 4x slower than an idle one, `chain-pc2-pv1c` against 4 October; the prover's cost to mining is 4%). A fleet of 18 mining 5090s, or 6 proving-only ones, covers an empty-block chain at 1 block/s through the chain mode; the mandatory rule waits for the measured share to reach one, not for these rows.
### The 12 GB requirement (Josh, 20:1xZ: "make sure we can prove on 12gb cards"): the GPU memory peak against SP1's knobs
Job `memsweep-pc2-pv1` (`tools/proving-v1/pc2-memory-sweep.ps1`), PC 2's RTX 5090 (32,607 MiB), the miners STOPPED by the job and the live prover switched off (its `sp1-gpu-server` would otherwise be the one the client connects to), every row: the server killed first, a 1-s `nvidia-smi memory.used` sampler, one `--mode compressed --shard 0` run of the pv1 host (`/opt/igneum-pv1`, SP1 6.8.1 cuda, `sp1-gpu-server` 6.8.1), 20:19 to 20:25Z. The knobs are the environment the GPU server inherits from the host process (`sp1-core-executor-6.8.1/src/opts.rs`: `SHARD_SIZE`, `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `MINIMAL_TRACE_CHUNK_THRESHOLD`, `TRACE_CHUNK_SLOTS`; `sp1-prover-6.8.1/src/worker/config.rs`: the `SP1_WORKER_NUM_*` and `*_BUFFER_SIZE` counts, defaults 4 core workers, 8 recursion prover workers). Idle card before the sweep: 1,732 MiB.
| Config (environment) | Fixture | Cycles | Peak MiB | Compressed prove s | Verified |
|---|---|---|---|---|---|
| baseline (no knob) | block-338-shard1, a full shard at `S_p` (6.75 M pgas) | 60,415,376 | **28,295** | 11.4 | yes |
| baseline | block-83616, an empty live shard | 280,706 | **13,863** | 2.3 | yes |
| ELEMENT_THRESHOLD 2^27 | full shard | 60.4 M | 28,326 | 10.9 | yes |
| ELEMENT_THRESHOLD 2^26, HEIGHT_THRESHOLD 2^21 | full shard | 60.4 M | 28,326 | 10.7 | yes |
| every worker count and buffer 1 | full shard | 60.4 M | 28,326 | 20.8 | yes |
| every worker count and buffer 2 | full shard | 60.4 M | 28,327 | 12.9 | yes |
| workers 1 + ELEMENT 2^27 | full shard | 60.4 M | 28,263 | 20.3 | yes |
| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 | full shard | 60.4 M | 28,326 | 20.6 | yes |
| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 + trace chunks 4 M x 2 slots | full shard | 60.4 M | 28,358 | 22.6 | yes |
| workers 1 + ELEMENT 2^25 + HEIGHT 2^20 | full shard | 60.4 M | 22,919 | 22.2 | yes |
| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 | empty shard | 280,706 | 13,861 | 2.6 | yes |
Reading. The GPU memory of a compressed shard proof is **13.9 GB for a shard of 280,000 cycles and 28.3 GB for one of 60 M cycles**, and no knob the environment carries moves the floor: the worker counts only slow the proof (11.4 s to 20.8 s), the trace thresholds at 2^26 and 2^27 change nothing, and the smallest trace threshold tried (2^25 elements, 2^20 rows) takes 5.4 GB off the full shard (22.9 GB) at twice the time. The floor sits in the GPU server's own allocation, not in the shard: an empty shard with every knob at its minimum still takes 13.9 GB. So on SP1 6.8.1's `sp1-gpu-server` as shipped, **a 12 GB card cannot prove even an empty shard** (13.9 GB), and the 11.0 GB target of tonight's requirement is out of reach from the environment. The S_p/2 and S_p/4 cuts of block 344 did not run: the package carries no `tools/prove-fixtures/seq.json` (the cut rows need the export; they would sit between the two measured points, and the floor is the binding number anyway). What is left to try, in order: the server's own options (its `--help` and the option names in its strings: the miner-on job prints them), SP1's core-only proof (the node needs the compressed proof, so this changes the protocol), and an SP1 release built for smaller cards (the 6.8.1 release notes are not read here; approximate: the project's documentation names 24 GB as the GPU requirement, `proving/windows-wsl2/setup-wsl.sh` quotes it).
### The same shard with the miner running (the 16 GB requirement), and the GPU server's own options
Job `memminer-pc2-pv1` (`tools/proving-v1/pc2-memory-miner-on.ps1`), 20:28 to 20:30Z, the miner at full rate on the card, the live prover off for the run, the same 1-s sampler: the full shard at `S_p` (60.4 M cycles) peaked at **30,039 MiB** and took 33.0 s (28,295 MiB and 11.4 s with the card to itself: the miner costs the prover 2.9x in time and 1.7 GB of memory); the empty shard **15,670 MiB** and 7.7 s (13,863 and 2.3 s alone). So a 32 GB card mines and proves the prototype shard with 2.5 GB to spare; a 24 GB card cannot prove it even alone (28.3 GB); a 16 GB card cannot hold even the empty shard beside the miner (15.7 GB, the display and driver on top). `sp1-gpu-server --help` prints only `--version`: it has no options of its own, and its strings carry no memory setting (`CUDA_OUT_OF_MEMORY` is an error name). The shard SIZE is therefore the only lever left on this build, measured next as the S_p curve.
The root-socket fault (the class, fixed the same evening). The chain and memory jobs ran the host as root inside WSL2; the first `sp1-gpu-server` they started left `/tmp/sp1-cuda-0.sock` owned by root, and the live prover (the app's own WSL user) then failed every shard with `CudaClientError: Connect(Os { code: 13, kind: PermissionDenied })` (PC 2 app log 1791230456, 20:00:56Z) until the socket was gone. Every pv1 playbook now kills the server and unlinks `/tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` fails CI on any playbook that runs a prove mode as root without both lines (shown failing on `pc2-prover-cost.ps1` before its `--mode id`-only exemption, passing after), and the plan carries the rule: a prover job on a shared card runs as the app's user or cleans its socket. It recurred at 21:25Z from another agent's job (agg-cost-pc2-1, the same root-run shape) and survived the 0.3.10 restart at 21:49:41Z; the fix job `socketfix-pc2-pv1` (`tools/proving-v1/pc2-socket-fix.ps1`, 22:01:14 to 22:02:12Z) found `/tmp/sp1-cuda-0.sock` owned by root, removed it, switched the prover off and on, and the app's next shard (block 89011 shard 0) was proven and submitted in 34 s and paid 0.93 IGN at 22:02:24Z. Playbooks that run the host: `tools/proving-v1/pc2-chain.ps1`, `pc2-memory-sweep.ps1`, `pc2-memory-miner-on.ps1`, `pc2-sp-curve.ps1` (all root, all with the cleanup now; the first two chain and sweep runs had none), `pc2-prover-cost.ps1` (`--mode id` only), `relay/playbooks/shard-test.ps1` and `proving/windows-wsl2/prove-shard.sh`, `prove-block.sh` (the app's user, not root), `tools/proving-v0/run.mjs` (the Mac, no server).
### The S_p curve: peak GPU memory against shard size against time, the card to itself (the first of the two curve jobs)
Job `spcurve-stopped-pc2-pv1` (`tools/proving-v1/pc2-sp-curve.ps1`, the miners stopped by the job, the live prover off, the server killed and its socket unlinked around every point, a 1-s `nvidia-smi` sampler), 20:33 to 20:37Z, PC 2's RTX 5090, the pv1 host (this run's `--budget` points were ignored by the pv1 host, so its block-344 rows are the fixture's own 6.75 M-pgas shard 0 twice; the pv1b host's re-plans at 2.25 M and 4.5 M pgas are the next job's rows). Idle card 1,743 MiB.
| Shard | pgas | Witness bytes | SP1 cycles | Peak MiB, card alone | Compressed prove s | Knob |
|---|---|---|---|---|---|---|
| block 83616, an empty live shard | 0 | 13,964 | 280,706 | **13,874** | 2.2 | none |
| block 56, one transfer | 600 | 4,902 | 556,369 | 13,907 | 3.2 | none |
| fees-v1-shards2 shard 0, a shard at the ADOPTED v1 budget (`S_p` 30,000; 4 transactions, 2 shards a block) | 22,172 | 18,390 | 4,717,439 | **20,434** | 4.3 | none |
| the same | 22,172 | 18,390 | 4.7 M | 20,435 | 3.7 | ELEMENT_THRESHOLD 2^25, HEIGHT 2^20 |
| block 338 shard 0, the full PROTOTYPE shard (`S_p` 7.5 M) | 6,751,568 | 21,611 | 60,415,376 | **28,307** | 10.8 | none |
| the same | 6.75 M | 21,611 | 60.4 M | 22,963 | 11.5 | ELEMENT_THRESHOLD 2^25, HEIGHT 2^20 |
| block 344 shard 0 (the fixture's own cut, 6.75 M pgas, modexp) | 6,748,392 | 18,535 | 59,678,420 | 28,275 and 28,307 | 11.5 and 11.0 | none |
The second job (`spcurve-stopped-pc2-pv1b`, the pv1b host whose `--budget` re-plans a fixture, 20:43 to 20:47Z, the same conditions) repeats the points (empty 13,875 MiB 2.1 s; one transfer 13,907 MiB 3.3 s; the v1 shard 20,435 MiB 4.2 s; the prototype shard 28,275 MiB 11.2 s) and adds the re-plans of block 344 (27 M pgas of modexp): at 2.25 M pgas (one transaction, 16 shards a block, 19,987,938 cycles) **28,371 MiB** and 6.6 s; at 4.5 M pgas (7 shards a block, 40,011,108 cycles) 28,307 MiB and 8.5 s; with the 2^25 trace threshold the 2.25 M shard 22,835 MiB and 6.2 s. So the peak is flat at 28.3 GB from 20 M cycles to 60 M (the server's buffers step up between 4.7 M and 20 M cycles and not after), and cutting the prototype shard smaller buys nothing until the v1 size.
The third job (`spcurve-miner-pc2-pv1`, the same points WITH THE MINER RUNNING on the card, 20:49Z on, the live prover off): empty shard 15,585 MiB and 7.5 s; one transfer 15,745 MiB and 12.7 s; **the v1 shard 22,210 MiB and 13.2 s** (20,435 and 4.2 s alone: the miner adds 1.8 GB and 3.1x); the 2.25 M shard 30,049 MiB and 17.9 s; the 4.5 M shard 29,954 MiB and 26.3 s; the prototype shard 30,083 MiB and 33.3 s. With the 2^25 trace threshold beside the miner: the 2.25 M shard 24,642 MiB and 21.5 s, the prototype shard 24,739 MiB and 38.8 s (24.7 GB: over a 24 GB card by the display's share, and 3.6x slower than the card alone). So beside the miner the adopted shard needs 22.2 GB: a 24 GB card (24,564 MiB) has 2.3 GB spare for it (the number for a 24 GB card is the 5090's allocation pattern on a 32 GB card, so approximate for the card itself), and the prototype shard needs 30.1 GB, the 32 GB card alone.
Reading, with the miner-on pairs above (empty shard 15,670 MiB, full prototype shard 30,039 MiB). The witness is never the binding term (4.9 to 21.6 KB a shard); the GPU server's working set is: a floor of 13.9 GB for any shard, 20.4 GB at 4.7 M cycles, 28.3 GB at 60 M cycles (23.0 GB with the smallest trace threshold, at the same time). By card: a **12 GB card proves nothing** on this build (the floor is 13.9 GB alone); a **16 GB card proves only empty and near-empty shards, alone** (13.9 GB; 15.7 GB beside the miner leaves nothing for the display); a **24 GB card proves the adopted v1 shard alone** (20.4 GB) and, at the miner's measured 1.7 GB extra, about 22.1 GB beside it (approximate: not measured on a 24 GB card), and never the prototype shard (28.3 GB); a **32 GB card proves the prototype shard beside the miner with 2.5 GB spare** (30.0 of 32.6 GB). The devnet is on the prototype table until H = 210,000 (6 October, about 19:50Z) and on the adopted v1 table (`S_p` 30,000 pgas) after it, so from H the 24 GB tier joins the provers and the shard that binds the memory is the 4.7 M-cycle one. Shards per block at each size: 1 at the prototype `S_p`, 4 at `B_p`; at the v1 budget 1 to 4 (one a block on tonight's chain, 2 to 3 on the txgen blocks).
### Step 4, the rule
| What | Measured |
|---|---|
| Unit tests | `cargo test --release -p kaspa-consensus-core -p igneum-exec --lib -- proving config::params::tests::override_params_carry_the_proving_v1 config::params::tests::consensus_digest` on this Mac (target `vendor/igneum-node/target-pv1`, 19:09Z): consensus core 13 passed (the segment record round trip, signature and the three nested sections; the credit split; the params switch and the digest that moves only once the switch is set), exec 8 passed (the segment grid and the split; the record checks: alignment, block, chain length, the veto naming the field, the deadline, the window; the chain rule both ways; the unproven restart; the shard side at 90%; the pool offering the segment section). The six full node suites go to PC 2 as a build job when the fleet is back |
| The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed. Run 3 on the FINAL fork tree (ece42979 on the 0.3.10 commit 21d4c73c, protocol 15, N = 8 both in the params default and `--segment 8`, the fast-time file's four fields), 20:52:41Z to 20:56:45Z: PASSED, 21 checks in 244.4 s (segments of 8: 119..126 paid in 1.0 s after submission, 127..134 refused fresh and paid continuing with chain_len 16, 135..142 left unproven and skipped, 143..150 restarted the chain) |
## 5 October 2026 (night), dp4a-class throughput on the M5 Max: the dot4 emulation against the ALU chain (Counter ASIC 2.0 layer 7)
Apple M5 Max, macOS 26, branch `ca2-analysis` (base `readwidth` 4badcee). The probes are standalone (no pack, no lottery kernel): `proto-metal/dot4-probe.swift` (built `swiftc -O -o dot4-probe dot4-probe.swift -framework Metal` under `with-lock.sh build`), `proto-opencl/dot4-probe.c` (built `cc -std=c99 -O2 -o dot4-probe-cl dot4-probe.c -framework OpenCL`), both run under `with-lock.sh measure` (exclusive; nothing else built or measured on the Mac during the runs). Shape: a dependent chain of one dot4 per step per lane, `acc = dot4(x, y, acc); x = x * 0x9E3779B1 + acc; y = rotl(y, 7) ^ (acc + s)`, 1,048,576 lanes x 4,096 steps, work-group 256, best of 3 with a fresh seed per repetition, device time (Metal: command buffer GPU start to end; OpenCL: event profiling). Beside it the ALU chain of the 9070 XT entry (`x = x * K + rotl(y, 7); y = (y ^ x) + s`, 5 ops per step counted). Every kernel is checked bit for bit against a CPU reference on lanes 0 and 1,048,575 in every repetition ("ok"). Design context: `docs/analysis/int8-matrix-family.md`.
| API, kernel | What one step is | best ms | G steps/s | ns per dependent step | ok |
|---|---|---|---|---|---|
| Metal, `probe_alu` | mul, add, rotate, xor, add | 4.882 | 879.8 (about 4.4 T int ops/s at 5 per step, approximate) | 1,192 | yes |
| Metal, `probe_dot4s` | signed dot4 emulated: `int4(as_type<char4>(a))` x same for b, 4 products summed into a wrapping int, plus the 3-op chain | 22.820 | 188.2 G dot4/s | 5,571 | yes |
| Metal, `probe_dot4u` | unsigned dot4 emulated: `uint4(as_type<uchar4>(a))`, same chain | 7.834 | 548.2 G dot4/s | 1,913 | yes |
| Apple OpenCL 1.2, `alu` | as Metal | 4.928 | 871.5 | 1,203 | yes |
| Apple OpenCL 1.2, `dot4e` | signed dot4 emulated with `convert_int4(as_char4(a))` | 22.797 | 188.4 G dot4/s | 5,566 | yes |
| Apple OpenCL 1.2, `dot4_khr` | `acc + dot(as_char4(x), as_char4(y))` under `#pragma OPENCL EXTENSION cl_khr_integer_dot_product : enable` | 5.076 | 846.2 | 1,239 | NO: mismatched the CPU reference on every lane checked in all 3 repetitions |
Reading: on this GPU a signed-byte dot4 costs 4.7 ALU-chain steps and an unsigned-byte one 1.6; Metal has no dp4a and no integer simdgroup matrix (MSL 4.1 sections 2.4 and 6.9), so these are the honest Apple costs of a per-lane dot4 family, and an unsigned definition is 3x cheaper for Apple at no cost to NVIDIA or AMD (both carry the unsigned form, PTX `dp4a.u32.u32`, AMD `v_dot4_u32_u8`). Apple's OpenCL does not list `cl_khr_integer_dot_product`; its `dot` on `char4` compiled anyway and returned something other than the integer dot (the mismatch), which is why a family's conformance vectors must gate every vendor path on the feature macro, not on "it compiled". Not run here: NVIDIA and AMD. The PC job is prepared and not published (coordinator's rule): `relay/playbooks/dot4-probe.ps1` with `dot4-probe-cl.exe` (proto-opencl/dot4-probe.c cross-compiled with mingw as `x86_64-w64-mingw32-gcc -std=c99 -O2 -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I proto-cuda/nvrtc/redist/include`, sha256 `5adaeb1aceb03dc41135baabe0b53f1ed5fac891a5b3c3849645b03efe4416f4`, 161,863 bytes); it runs the scalar, KHR, AMD `__builtin_amdgcn_sudot4` and NVIDIA inline-PTX `dp4a` variants on every OpenCL GPU of the machine with the mining cards switched off through `/api/cards` and restored after. The CUDA form (`proto-cuda/dot4-probe.cu`, `__dp4a`) needs nvcc on the PC and is the cross-check.
**PC 1, 5 October 2026 20:29 UTC, the same probe on the RTX 5090 and the RX 9070 XT** (machine ae432dc7, Windows 11; fetch job `fetch-dot4-20261005` placed `dot4-probe-cl.exe` sha256 `5adaeb1a…6416f4`, run job `run-dot4-20261005` ran `relay/playbooks/dot4-probe.ps1`: the app's `nvidia:0` and `amd:1:gfx1201` cards switched off through `POST api/cards`, the probe run on every OpenCL device, the cards restored with their settings (identities 8 and 2, power cap 80% and none); `node tools/jobs.mjs run-dot4-20261005`; 101 s wall, every kernel under 10 ms; device event time, best of 3, same lanes and steps as the Mac rows):
| Device, platform | alu, G steps/s (ms) | dot4e signed emulation, G dot4/s (ms) | dot4 instruction, G dot4/s (ms) | `cl_khr_integer_dot_product` | ok |
|---|---|---|---|---|---|
| RTX 5090, NVIDIA OpenCL 3.0 CUDA, driver 617.14 | 8,753.5 (0.491) | 1,239.1 (3.466), 7.1x the ALU step | 7,453.6 (0.576) via inline PTX `dp4a.s32.s32`, 1.17x the ALU step | not listed; the `dot(char4,char4)` kernel does not build | yes |
| RX 9070 XT (gfx1201), AMD-APP 3683.0 (PAL,LC), OpenCL 2.0 | 701.4 (6.124) | 480.8 (8.932), 1.46x | 664.3 (6.465) via `__builtin_amdgcn_sudot4`, 1.06x | not listed; same | yes |
| RX 9070 XT, the older 3652.0 platform entry (dup) | 696.2 (6.169) | 501.7 (8.561) | 683.6 (6.283) | not listed | yes |
| gfx1036 (integrated RDNA 2, 2 CUs), 3683.0 | 40.6 (105.9) | 15.8 (272.3), 2.6x | `sudot4` does not build: "needs target feature dot8-insts" | not listed | alu and dot4e yes |
Reading: one `dp4a` on the 5090 costs about one ALU-chain step (7.45 T dot4/s, 0.85 of the chain's 8.75 T steps/s); one `v_dot4_i32_iu8` on the 9070 XT the same (0.66 T, 0.95 of its chain). Emulating the signed dot4 costs 6.0x the instruction on NVIDIA (the OpenCL compiler does not fold the four sign-extended products into `dp4a`) and 1.38x on AMD. Vendor ratios: the 5090 is 12.5x the 9070 XT on the ALU chain and 11.2x on hardware dot4; against the M5 Max's best (unsigned emulation, 0.55 T) it is 10x on the chain and 13.6x on dot4. The hash itself is bound by DRAM reads, so these per-op numbers bound a family's cost and are not hash rates (`docs/analysis/int8-matrix-family.md` section 4). Adrenalin's OpenCL C accepts the clang builtin and emits the instruction on RDNA 4 (the third-party RDNA 3 report of the same route is now confirmed on this card); no PC platform lists the Khronos integer-dot extension. The 5090 SM clock read 2,505 MHz before and after (nvidia-smi; 2,850 MHz while mining in the telemetry entry), so the card was idle for the probe.
## 5 October 2026, layer 3 scratch soundness (Counter ASIC 2.0 step 4; branch ca2-soundness on readwidth b970dda; cryptographer)
Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0, other agents' builds and the readwidth measurements running beside (the Mac measure lock was free during the GPU runs; nothing here is a hash-rate figure). Write-up `docs/analysis/scratch-soundness.md`; tests `igneum-pow/tests/scratch.rs`; harness `proto-metal/packbench` built from this branch (`--batch-base` added) into the session scratchpad with `swiftc -O -target arm64-apple-macos11 -framework Metal`.
CPU, `with-lock.sh build nice -n 19 ~/.cargo/bin/cargo test -j4 --test scratch -- --nocapture` (3.6 s): 7 of 7 pass. Stats, 6 classes x 3 seeds x 2^11 units, every read-modify-write traced (3.1 to 12.6 million per class): written-word bias within 6 sigma (worst 3.63); re-hit rate measured against the uniform birthday rate 12.58 vs 10.91 percent (scr2k32), 21.47 vs 20.83 (scr4k32), 36.99 vs 36.50 (scr8k32), 3.84 vs 2.88 (scr2k128), 6.25 vs 5.82 (scr4k128), 11.89 vs 11.37 (scr8k128); slot histogram non-uniform (hottest slot 1.39x to 5.10x the mean: the slot is a register's low bits); deepest chain 5 to 9. Edge: 7 hand-built programs x 2 geometries x 4 bases against an independent hand model, 56 of 56, and 56 of 56 mismatches with the hand model's rewrite words swapped. Static scratch check: 42 of 42 emitted kernels of the 7 scr packs (regenerated byte for byte from program.json first), 6 deliberate breaks caught. Fuzz: 200 generated scratch programs, contract and acceptance on every instruction, 800 units; `IGNEUM_SCRATCH_PACKS_OUT` wrote 214 packs (57 s, three memory-hard caches). The crate's other tests: 33 of 34 lib tests pass; `verify::tests::fold_and_wide_fetch` fails on the readwidth tip itself (`verify.rs:508`, `k as u32 * 0x9E37_79B1` overflows under the test profile; not touched here).
Metal, `with-lock.sh run <script>`, scripts `gpu-a.sh` and `gpu-b.sh` in the session scratchpad (one `packbench` call per line):
| Run | Command shape | Result |
|---|---|---|
| scr4k32 standard pack, timing | `packbench --pack proto-cuda/packs-readwidth/scr4k32 --batches 1 --batch-log2 24 --warps 2048` | 3/3 standalone, 3/3 in batch, fingerprint `3d1af881bd978fb9`, 1.8 s wall for the run |
| warp-count independence | same pack, `--batch-log2 12 --warps 1`, `2`, `128` | fingerprint `8c07620f4d9adefd` at all three |
| wrap inside the launch | `--batch-log2 9 --batch-base 4294967040 --warps 1`, `4` | fingerprint `8e9e233234d3a297` at both, the base-0 vector inside the window after the wrap 1/1 |
| broken tag (`tag = salt`) on a copy of scr4k32, standard vectors | `--batch-log2 24 --warps 2048` | standalone 3/3, in batch 2/3 (base 1,000,000, warp 530's 16th unit, caught), overall FAIL |
| broken tag, 2 units on 1 warp, standard vectors | `--batch-log2 6 --warps 1` | 3/3, 1/1, PASS: missed, the standard vectors have no base 32 |
| 14 edge packs, run A | `--batch-log2 6 --warps 1 --batches 1` | 14/14 PASS, 4/4 standalone and 2/2 in batch each (bases 0 and 32 on one arena) |
| 14 edge packs, run B | `--batch-log2 9 --warps 1 --batch-base 4294967040` | 14/14 PASS, 4/4 and 3/3 each (16 units on one arena, the wrap inside) |
| broken tag on edge-slot0 at 32 and 128 KiB | `--batch-log2 6 --warps 1` | 4/4 standalone, 1/2 in batch, FAIL at both (the second unit read the first's slot 0) |
| broken lazy fill (`m_` all ones) on edge-slot0 at 32 KiB | same | 0/4, 0/2, FAIL |
| 200 fuzz packs | `--batch-log2 9 --warps 2 --batch-base 4294967040 --batches 1` each | 200/200 PASS, 800/800 standalone units (25,600 hashes), 400/400 in the window; 91 s wall for the 200 runs (20:16:02 to 20:17:33 UTC) |
Totals: 228 of 228 PASS where expected, 3 of 3 FAIL where built in. Reading: the one-warp CPU simulation is exact on Metal under the hosts' present tag policy; the analysis names the host contract (zero the arena at allocation and at the tag counter's wrap, tags from 1, groups a multiple of warps) that turns that into a guarantee, and finds layer 3 does not move the named chip (section 3.4 of the write-up: 2.4x at every share under the cap).
## 5 October 2026 (night), epoch length as an era parameter (Counter ASIC 2.0, layer 9): the Mac's compile-ahead per program
Branch `ca2-epoch`, worker "ca2-epoch"; design and the per-card table in `docs/plans/epoch-length.md`. Question (Josh: "what about faster program changes?"): what a card spends per epoch between receiving the next seed and swapping, which sets the floor of the epoch-length ladder (600 to 7,200 DAA s). Machine: Apple M5 Max (Darwin 25.6.0, 64 GiB), 21:18 UTC, load average 11 to 14 from other agents' builds and runs; the measure lock held for the 3-s run (`tools/lock/with-lock.sh measure bash scratchpad/epoch-measure.sh`). `proto-metal/igneum-bench` built from this branch with `swiftc -O -target arm64-apple-macos11 -o igneum-bench main.swift -framework Metal` under a build slot.
Ten distinct programs (seed strings `igneum-devnet-v4-epoch0`, `/epoch1` .. `/epoch9`; version 2 generator, 128 loads per hash), each generated and compiled at run time (`makeLibrary` from source plus `makeComputePipelineState`), dataset 2^28 words, one 2^20 batch and one verify warp per program:
./igneum-bench --seed igneum-devnet-v4-epoch0 --hours 10 --dataset-log2 28 --batch-log2 20 --batches 1 --verify-warps 1
| Program | Compile ms (library + pipeline) | Mhash/s (GPU) | Verify |
|---|---|---|---|
| epoch0 | 18.8 (9.3 + 9.5) | 27.2 | PASS |
| epoch1 | 17.8 (8.8 + 9.1) | 27.9 | PASS |
| epoch2 | 17.6 (8.6 + 9.1) | 27.8 | PASS |
| epoch3 | 16.1 (8.0 + 8.2) | 27.7 | PASS |
| epoch4 | 18.6 (9.0 + 9.6) | 27.5 | PASS |
| epoch5 | 15.9 (7.7 + 8.2) | 28.4 | PASS |
| epoch6 | 17.6 (8.6 + 9.1) | 27.8 | PASS |
| epoch7 | 17.6 (8.7 + 9.0) | 30.1 | PASS |
| epoch8 | 20.4 (9.8 + 10.6) | 28.4 | PASS |
| epoch9 | 18.2 (8.7 + 9.5) | 29.3 | PASS |
| min / median / max | 15.9 / 17.7 / 20.4 | | 10 of 10 |
Cache fill 1.95 ms GPU (192.4 ms one core), dataset build 20.8 ms GPU for 1 GiB. The devnet pack three times through `packbench --pack ../proto-cuda/packs/igneum-devnet-v4-epoch0 --batches 1 --batch-log2 20 --group 256` (the pack's two libraries, `memhard.metal` and `program.metal`): compile 79 ms, 1 ms, 1 ms (the system shader cache answers the identical source from the second run); cache fill 0.6 to 0.7 ms GPU, dataset build 20.7 to 20.8 ms GPU.
Reading: a fresh program compiles in about 18 ms on this card with the Metal compiler service warm, 79 ms for a pack with its dataset kernels, up to 1.8 s cold (the variant-racing entry's first seed), 0 to 444 ms at the fleet's live boundaries (M11). The hot table fill of layer 5 is 0.07 to 0.22 ms (ca2-cache). So the Mac's per-epoch compile-ahead is under 2 s without the race and about 38 s with it (M11: 34.0 / 34.9 / 37.8 s), and the race is the only item visible against the 600-s window in which the program is known (lead 1,200 s minus the 600-s VDF, fixed at every epoch length). PC cards, cited in the plan: RTX 5090 NVRTC 151 to 180 ms, prepare 0.5 to 1.0 s without the dataset (M11), race one round about 37 s; RX 9070 XT OpenCL compile NOT MEASURED at the current worker (owed: `host.c` times `clBuildProgram` only in the `prepare` path and no `prepared` line from gfx1201 is in any upload); Intel UHD build 3.0 to 6.4 s (M11). Floor by the rule (slowest compile-ahead under 10% of the epoch and inside the window, dataset excluded): 600 DAA s, carried by the race at 6.3% of 600 s; with the race off (M11 found base wins on both the 5090 and the Mac) the slowest measured row is the Intel iGPU at 1.1%. Consequences per tier and the difficulty-settle constraint (24% of a 600-s epoch in settle at the measured 144 s) are in the plan.

View file

@ -39,11 +39,11 @@ Versions in the table: `igneum-pow` is the Rust crate at `igneum-pow/Cargo.toml`
| 12 | The difficulty rule recovers from a hashrate step within minutes, where Kaspa's sampled rule never settles. A step inside an epoch set the rule oscillating on the live devnet on 4 October 2026; rule v2 removes it in the simulator and on a test network and is built but not yet rolled out | Spec 2.3; litepaper Speed (implied); bench page | tested by the team | repo `e9328c6`, `abb5a5d` (attacks), `67bf226` (rule v2); fork `difficulty` branch (timestamp fix) and `devnet-v4` `a21ff239` (`difficulty_v2_activation_daa`, `REF_WINDOW_V2 = 600`); `sim/difficulty/sim.py --live` | The live record `sim/difficulty/records/live-2026-10-04.csv` (8,090 headers, `pull_live.py`) and the hash-rate record beside it; `sim/difficulty/sim.py` on the synthetic set and the DAG replay; `sim/difficulty/attacks/attacks.py`; `sim/difficulty/testnet_v2.py` (3 nodes, activation at DAA 900); `cargo test --release -p kaspa-consensus --lib difficulty` (15 pass); bench-log "difficulty controller", "difficulty rule under attack", "timestamp attack fixed", "difficulty rule v2" | Live devnet v4, 4 October 2026 (UTC): a second RTX 5090 joining 7 minutes into an epoch (about 152 to 280 MH/s) hardened the difficulty 70M to 144M in 90 s and then swung by about a third for 40 minutes around the true level of 139M while the epoch-long reference lane carried the join; that card leaving for 4 minutes eased 116M to 67M and back to 106M; the epoch boundary with both PCs restarting took 152M to 77M in 3 minutes, after which the rule held within 1.3% per minute with no flips. Cause: the reference lane covered the whole epoch, so a mid-epoch step polluted it for the hour and the 25% trigger flipped on the short lane's noise. The DAG replay reproduces the record (std of log difficulty 0.115 against 0.134, 4.3 peaks against 4). Rule v2 (reference window 600 DAA) on the replay: std 0.026, 0 flips, mean 142.6M against 139M true; on a 3-node test network the v2 nodes eased a leave with no peak and held a rejoin within 3% after 60 s, and a node without the activation height forked off at it as designed. Rule v2 rolled onto the 12-node cloud network on 4 October (all nodes crossed the height on one chain; a hash-rate step then settled in 160 to 270 s with no swing) and activates on the devnet at DAA 33,000 the same evening. Timestamp forging (ledger M23) fixed the same day: a 50% forger drifts the rate under 1.1% where the 3 October rule gave it a 9.9x difficulty. Simulator, settled seconds: x50 step 62 to 66 (Kaspa 1,542), /50 step 657 to 753 (Kaspa 12,296). Apple M5 Max under load 7 to 442; the DAG model is fitted on one scale; the pool hopper's 0.7-point excess over Kaspa's rule stays open | none yet |
| 13 | Every node executes the ordered transactions natively and reaches the same state root | Litepaper Proving ("Every node executes ... natively"), Building ("runs on Igneum unchanged") | tested by the team | repo `f5f8c80`, `8dae48b`; fork `devnet-v4` `dc749905`; revm 43.0.3 | `node tools/evm-smoke/smoke.mjs` against a 3-node `igneumd`; `igneum-exec-diff seq.json`; bench-log "execution layer devnet v3" and "devnet-v4 integration" | Simnet, 3 October 2026: 87 of 87 viem checks, state roots identical on 3 nodes at four heights, 57 executed and 19 skipped transactions agree with plain revm, 0 mismatches. Merged node on real proof of work, 4 October 2026: 84 of 85 checks (the miss needs parallel blocks the network did not produce in 36 s), 59 transfers in 10 chain blocks, state roots identical on 3 nodes, `igneum-exec-diff` 0 mismatches over 59 transactions; the live devnet v4 runs this execution layer. Apple M5 Max. The prover is a stub; state is rebuilt from genesis at start; no EVM transaction relay between nodes | none yet |
| 14 | Ethereum bytecode runs unchanged, with the documented differences of spec 7.1 | Homepage Build card; litepaper Building | tested by the team | as row 13; fixes `F-exec-A`, `F-exec-B` (spec 7.5) | `tools/evm-smoke/smoke.mjs`: deploy via viem, `increment`, `hashLoop`, `eth_estimateGas`, `eth_getLogs`; `tools/exec-attacks` scenarios 1 and 3; bench-log "execution layer attack fixes" | Deployment, calls, reverts, logs and gas estimates behave as viem expects; chain id 4463; the prototype pgas table gives 0.0095 to 0.028 pgas per gas, below the design's band before calibration, 3 October 2026. 4 October 2026: a transaction that would cross the block's proving budget is refused by the mempool and, if forced in, aborted and charged with its nonce advanced (25 of 25 checks; 30 of 30 malformed cases). Apple M5 Max. The `Prover` precompile, proof records and the shard planner are not in the node | none yet |
| 15 | Every block is proven, with the proof landing within about a minute at launch | Homepage stats ("~60 s to a proof"); litepaper Proving; roadmap phase 3 gate | implemented | repo `d7e1f89` (GPU proof), `e01a3cc`, `292e800`, `eedd136` (`proving/igneum-prove`: shard cutter, MPT witnesses, shard and aggregator guests); SP1 6.8.1; spec 7.2, 7.6 | `proving/windows-wsl2` (SETUP-PROVER, PROVE-BLOCK) on the RTX 5090; `igneum-prove-host --mode block` on `proving/fixtures/`; bench-log "proving v0 on the RTX 5090" and "proving: devnet v4 shards" | First GPU proof of an Igneum block, 4 October 2026, RTX 5090 (WSL2, SP1 cuda, mining paused): fixture `block-78-increment` (2 transactions), core proof 1.4 s (7.3 MB, verify 0.221 s), compressed proof 2.7 s (1.27 MB, verify 0.038 s), post-state and receipts roots identical to the node's; 15.7x and 20.6x faster than a loaded M5 Max CPU. The same day on that CPU (load 38 to 47): a three-shard block proved shard by shard and aggregated by recursion, 19 min (1,139 s) end to end, 245 to 337 s per compressed shard proof, every proof verified. What is not there: no proof is produced, carried or checked on the chain (the devnet prover is a stub that signs claims), the proving pool pays nobody (row 21), the block proven is far below one shard, and the 60-second figure remains a design target; the pass mark is the standard in `docs/benchmarks/proving-e2e.md`. Second RTX 5090 run, 4 October 2026 evening (job run-20261004-173115): a full shard at the provisional S_p (6.75 M pgas, 60.8 M cycles) executed in 1.63 s, core proof 8.3 s (18.1 MB), compressed proof 10.9 s (1.27 MB, verify 0.040 s); a two-shard block (13.5 M pgas) proved shard by shard (11.7 s and 10.0 s) and aggregated in 2.2 s, 24 s of GPU stages end to end, every proof verified, six tampered witnesses rejected. The two host defects (an abort after the upload, an idle wait that turned out to be an unbuffered 18 MB proof save through the WSL2 file bridge, 24 minutes) are fixed (ledger P20) 5 October 2026, live devnet with real transactions (bench-log "real transactions, the first non-empty shard proven and paid"): block 72704 shard 0, 29 transfers, 5,800 pgas, proven on PC 2 in 34 s, verified on the Mac in 0.297 s and paid 1.7623 IGN, 53 s after the chain block executed; of about 1,400 blocks in the 20-minute window 36 were proven (the one prover takes the newest shard assigned to it), so "every block" is not yet true; a second content shard (72803, all copies skipped) failed the native-execution veto on the exporter's block structure, fixed with fixtures the same day, the node side pending the 0.3.9 rollout | none yet |
| 16 | A 12 GB card proves one shard in about 20 s | Litepaper Proving ("The proving budget"); roadmap gate 2 | designed | spec 5.1 (Target), 7.6 (`S_p` provisional, 7,500,000 pgas = `B_p` / 4) | `PROVE-SHARD.bat` on the RTX 5090 (pending); the end-to-end standard in `docs/benchmarks/proving-e2e.md`; bench-log "proving: devnet v4 shards" | Measured on a 32 GB card, not yet on a 12 GB card. A shard at the provisional `S_p` is 60.8 M SP1 cycles on the prototype pgas table (9 cycles per pgas, 44 per EVM gas; the modexp entry about 100x its SP1 cost); on an RTX 5090 (4 October 2026 evening, job run-20261004-173115) it executed in 1.63 s and its compressed proof took 10.9 s, verified in 0.040 s, so the 32 GB card is inside the 20 s target with margin. Whether a 12 GB card proves it at all, and in what time, is the next measurement (an RTX 3060 and an RTX 5060 Ti 16 GB are on order). A per-shard time can be met by shrinking the shard, so the project does not use it as a pass mark | none yet |
| 17 | The chip resistance target: a chip gains under 2x over a GPU | Litepaper Mining, "What Igneum does not claim"; homepage "no chip can be built for it" | designed | spec 0.2 (Target); O-1.17 | Public benchmark with a leaderboard by card model and a standing bounty, January 2027 (O-1.17); the on-die-SRAM test on the RTX 5090 (R3.5) | A target, not a measurement. Review round 3 priced a recompute chip with the 256 MiB cache on die at about 2.4x, approximate, before the usual chip-versus-GPU integer gain; the design answer (cache larger than any die) is open (spec 1.16) | none yet |
| 18 | The chip resistance measurements: the program is random-access bound, not bandwidth bound, and sits beyond a card's on-chip cache | Litepaper Mining ("bound by memory bandwidth", to be corrected), vs RandomX "Measured so far" | tested by the team | repo `aba248d`, `f2a1a64`, `4b95c5e` | RTX 5090 dataset sweep 4 MiB to 1 GiB with `proto-cuda/host.cu`; bench-log "RTX 5090 first run" and "dataset sweep" | At 1 GiB: 228.1 Mhash/s, 23.7 G random loads/s, 94.9 GB/s useful against a 1,638 GB/s dataset fill; inside the 96 MiB L2 (4 and 64 MiB) 1,340 to 1,353 Mhash/s, about 5.8x faster; 104 against 128 loads per hash gives 228 against 185 Mhash/s, proportional. 3 October 2026, RTX 5090, Windows, CUDA 12.8, version 1 programs. Prototype dataset 1 GiB against 2 GB at genesis; a pure random-read microbenchmark (R3 chip designer, attack 2) has not run; the sweep has not been repeated on version 2 | none yet |
| 19 | The lottery hash is sound as a hash: uniform output, deterministic, no out-of-bounds read, fuzzed | Litepaper vs RandomX ("Every number above is measured and logged") | tested by the team | repo `c52307e`, `58a5a63`, `b27da39`; `proto-metal/TESTS.md` | `proto-metal/igneum-bench --fuzz --edge --stats --determinism --memcheck`; `--fuzz 2000` on the version 2 generator; `igneum-census`; bench-log "hardening tests", the re-run on the memory-hard dataset, "generator version 2 adopted" | Version 1: 10,200 random programs, 1,305,600 hashes, 0 mismatches; 14 of 14 edge cases; bit frequency within 2.90 sigma, avalanche mean 31.99 to 32.04 of 32; deterministic fingerprint across 5 runs; every dataset read masked, 3 October 2026. Version 2, 4 October 2026: 2,000 random programs through the Metal cross-check, 8,000 warps, 0 mismatches, 128 loads per hash on every program; 20,000-program census, 5.2% rejected (4.1% static, 1.1% dynamic). Apple M5 Max. Statistics are not a security proof; the edge, stats and memcheck sections were not re-run on version 2 (they do not depend on the generator); the seed derivation review (O-1.4) is open; the fuzz set has run on Metal and the CPU only | none yet |
| 15 | Every block is proven, with the proof landing within about a minute at launch | Homepage stats ("~60 s to a proof"); litepaper Proving; roadmap phase 3 gate | implemented | repo `d7e1f89` (GPU proof), `e01a3cc`, `292e800`, `eedd136` (`proving/igneum-prove`: shard cutter, MPT witnesses, shard and aggregator guests); SP1 6.8.1; spec 7.2, 7.6 | `proving/windows-wsl2` (SETUP-PROVER, PROVE-BLOCK) on the RTX 5090; `igneum-prove-host --mode block` on `proving/fixtures/`; bench-log "proving v0 on the RTX 5090" and "proving: devnet v4 shards" | First GPU proof of an Igneum block, 4 October 2026, RTX 5090 (WSL2, SP1 cuda, mining paused): fixture `block-78-increment` (2 transactions), core proof 1.4 s (7.3 MB, verify 0.221 s), compressed proof 2.7 s (1.27 MB, verify 0.038 s), post-state and receipts roots identical to the node's; 15.7x and 20.6x faster than a loaded M5 Max CPU. The same day on that CPU (load 38 to 47): a three-shard block proved shard by shard and aggregated by recursion, 19 min (1,139 s) end to end, 245 to 337 s per compressed shard proof, every proof verified. What is not there: no proof is produced, carried or checked on the chain (the devnet prover is a stub that signs claims), the proving pool pays nobody (row 21), the block proven is far below one shard, and the 60-second figure remains a design target; the pass mark is the standard in `docs/benchmarks/proving-e2e.md`. Second RTX 5090 run, 4 October 2026 evening (job run-20261004-173115): a full shard at the provisional S_p (6.75 M pgas, 60.8 M cycles) executed in 1.63 s, core proof 8.3 s (18.1 MB), compressed proof 10.9 s (1.27 MB, verify 0.040 s); a two-shard block (13.5 M pgas) proved shard by shard (11.7 s and 10.0 s) and aggregated in 2.2 s, 24 s of GPU stages end to end, every proof verified, six tampered witnesses rejected. The two host defects (an abort after the upload, an idle wait that turned out to be an unbuffered 18 MB proof save through the WSL2 file bridge, 24 minutes) are fixed (ledger P20) 5 October 2026, live devnet with real transactions (bench-log "real transactions, the first non-empty shard proven and paid"): block 72704 shard 0, 29 transfers, 5,800 pgas, proven on PC 2 in 34 s, verified on the Mac in 0.297 s and paid 1.7623 IGN, 53 s after the chain block executed; of about 1,400 blocks in the 20-minute window 36 were proven (the one prover takes the newest shard assigned to it), so "every block" is not yet true; a second content shard (72803, all copies skipped) failed the native-execution veto on the exporter's block structure, fixed with fixtures the same day, the node side pending the 0.3.9 rollout 5 October 2026, evening (bench-log "proving v1"): the aggregated segment record, the chain rule and the unproven rule are implemented behind `proving_v1_activation_daa` (branch proving-v1, not on the devnet before 0.3.11); on the RTX 5090 a chain of 8 consecutive live blocks proved and aggregated by recursion in 135.6 s with the miner on the card (17 s a block, one proof of 1,272,909 bytes attesting all 8, verified in 0.04 s); the 3-node fast-time harness paid a segment record 1.0 s after submission and refused a late one after its deadline (21 checks); the devnet itself, with one prover, carried proofs for 2.4% of blocks over 30 minutes at a block-to-record latency p50 44 s, p99 52 s. The "within about a minute" holds per proven block; "every block" needs 18 mining 5090s or 6 proving-only cards at empty blocks on the measured rates, and the mandatory rule stays off until the share is one | none yet |
| 16 | A 12 GB card proves one shard in about 20 s (WITHDRAWN 5 October 2026: a 24 GB card proves a full shard at the adopted size in 4.3 s; 32 GB mines and proves) | Litepaper Proving ("The proving budget"); roadmap gate 2 | designed | spec 5.1 (Target), 7.6 (`S_p` provisional, 7,500,000 pgas = `B_p` / 4) | `PROVE-SHARD.bat` on the RTX 5090 (pending); the end-to-end standard in `docs/benchmarks/proving-e2e.md`; bench-log "proving: devnet v4 shards" | Measured on a 32 GB card, not yet on a 12 GB card. A shard at the provisional `S_p` is 60.8 M SP1 cycles on the prototype pgas table (9 cycles per pgas, 44 per EVM gas; the modexp entry about 100x its SP1 cost); on an RTX 5090 (4 October 2026 evening, job run-20261004-173115) it executed in 1.63 s and its compressed proof took 10.9 s, verified in 0.040 s, so the 32 GB card is inside the 20 s target with margin. Whether a 12 GB card proves it at all, and in what time, is the next measurement (an RTX 3060 and an RTX 5060 Ti 16 GB are on order). A per-shard time can be met by shrinking the shard, so the project does not use it as a pass mark 5 October 2026, evening (bench-log "proving v1", the S_p curve): measured on the RTX 5090 with SP1 6.8.1's GPU prover, the card to itself, 1-s nvidia-smi samples: an empty shard 13,874 MiB and 2.2 s; a full shard at the ADOPTED v1 budget (30,000 pgas, 4.7 M cycles) 20,434 MiB and 4.3 s; the full prototype shard (6.75 M pgas, 60 M cycles) 28,307 MiB and 10.8 s; beside the miner 15,670 and 30,039 MiB. No environment knob of SP1 moves the 13.9 GB floor and the GPU server has no options of its own, so on this build a 12 GB card proves nothing, a 16 GB card only empty shards, a 24 GB card the adopted full shard alone and beside the miner (22,210 MiB and 13.2 s, measured on the 32 GB card: the 5090's allocation pattern, not yet a run on a 24 GB card) and a 32 GB card the prototype shard beside the miner with 2.5 GB spare. The litepaper line now says so; the 12 GB gate returns when a prover build with a smaller floor is measured on a 12 GB card | none yet |
| 17 | The chip resistance target: a chip gains under 2x over a GPU | Homepage hero and litepaper abstract ("a custom chip gains under 2x, and the model and the bounty are public"), litepaper "What Igneum does not claim" | tested by the team (the model), designed (the target) | program class v3 (Counter ASIC 2.0, 5 October 2026): branches ca2-v3 d233fa1 and after, ca2-mixer 1ab8b21, ca2-era 78c0ee4; `docs/analysis/chip-model-v3.md`, `docs/analysis/sram-mirror.md`, `docs/analysis/scratch-soundness.md` | The m16 recompute model re-run on the measured v3 rates and verifier times; the on-die-cache chip row | The on-die-cache recompute chip against the RTX 5090's measured 136.1 MH/s: class v2 2.4x; class v3 (mixer x8) 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon; margin 8% on the allowance, 9% on the budget. 5 October 2026, M5 Max, RTX 5090, RX 9070 XT. The 2x target is a target: no chip has been built; the bounty stands (O-1.17) | none yet |
| 18 | The chip resistance measurements: the program is latency-bound (random reads), not bandwidth-bound, on every card we own, and sits beyond a card's on-chip cache | Litepaper Mining ("waits on memory latency, not on maths or bandwidth"), vs RandomX; the numbers page | tested by the team | readwidth e752fc7 (`docs/plans/read-width.md`), ca2-era 78c0ee4, ca2-cache 2de19e5 (`docs/plans/hot-table.md`) | The dependent-read probes at 32 to 1,024 MiB and the hash rate per class on the three cards; the latency-bound share = rate over the probe ceiling per load | Latency-bound share at the 1 GiB dataset: RTX 5090 0.96 (v2) and 1.01 (v3), RX 9070 XT 0.87 and 0.95, M5 Max 1.01 and 1.06; wider reads do not close the AMD gap (the 9070 XT does 2.4 G dependent reads per second at every width; the 5090 goes bandwidth-bound at 64 B, share 0.58); a 32 to 96 MiB hot table is not kept resident by any card while the dataset streams (g 0.80 to 0.87 in the added form). 5 October 2026 | none yet |
| 19 | The lottery hash is sound as a hash: uniform output, deterministic, no out-of-bounds read, fuzzed; class v3 bit-exact on the three vendors | Litepaper vs RandomX ("Every number above is measured and logged"), the numbers page | tested by the team | ca2-mixer 1ab8b21 (`tests/mixer.rs`, `tests/scratch.rs`), ca2-era 78c0ee4, ca2-soundness a465881 (`docs/analysis/scratch-soundness.md`), `igneum-pow/tests/packs.rs` | The crate suite (53 + 4 + 19 + 7), the Metal fuzz, edge, stats and determinism runs on the v3 construction, the pack vectors and 2^24 fingerprints on Metal, Apple OpenCL, the RTX 5090 and the RX 9070 XT, the 1,024-hash CPU re-check per card | Class v3 (mixer x8 + era): 200-program fuzz 200 of 200 on Metal, every tenth on Apple OpenCL; the pinned v3 packs 3/3 + 3/3 and 96 of 96 lanes on Metal and Apple OpenCL; the six era packs' fingerprints equal on the three vendors (PC 1 job run-ca2-era-pc1-20261005, 5 October 2026); the v2 exports byte-identical on the v3 crate; the final-class PC rows and the G2 re-check: job run-ca2-era-pc1b-20261005 (pending at the time of writing) | none yet |
| 20 | No premine, no pre-sale, no allocation: every coin is minted by the schedule and every coin goes to the block producer (80%) and the proving pool (20%) | Homepage stats and Economics tiles; litepaper Supply, Economics | implemented | repo `6ac80a3`; fork "igneum-node devnet v0"; `consensus/core/src/igneum.rs`, `coinbase.rs` | `cargo test -p kaspa-consensus-core igneum` (8 pass: subsidy table, ramp, split, cap) and `cargo test -p kaspa-consensus coinbase` (8 pass); `igneum-miner inspect 40`; bench-log "igneum-node devnet v0" | Coinbases on the devnet: 80/20 exact on 39 of 39 single-payee blocks, the 20% to the `igneum-proving-pool-v0` output; the per-second schedule sums to under the 4,000,000,000 cap by less than 100 coins; 3,168,808,781 units per DAA second in years 0 to 2, halving at 63,115,200 DAA s. 3 October 2026, Apple M5 Max. The devnet genesis carries no allocation; the mainnet genesis does not exist yet, so the claim is about the code and the stated rule, not a launch that has happened | none yet |
| 21 | The proving pool's 20% reaches shard provers and aggregators | Litepaper Economics; homepage "20% provers" | tested by the team | spec 5.3; `proving/igneum-prove` carries the prover's payout address in every shard proof (ledger P12) | None. The pool output exists (row 20); the payout from it against proof records is unwritten. Since 5 October 2026: the payout rule is live on the devnet (`proving.rs shard_payouts`, the carrying segment pays the first valid record per shard its part of the segment's pool credit) | The escrow accumulated on the simnet (92.55 IGN at the end of the v3 run) and nothing can draw it. Rule decided: per block, divided among shards by consensus proving cost, sortition to 8 provers for 10 s then open (spec 7.2). The economy model of 4 October 2026 (`sim/economy`, 1,000 operators, 30 days) kept every block proven within 60 s under six stress scenarios; a model, not hardware Live devnet, 5 October 2026: 388 shards paid by 16:02 UTC, 446.13 IGN from the pool to PC 2's payout address, 0.8813 IGN per mergeset block of the proven segment (bench-log entries of 5 October: "the first shards proven, verified and paid" and "real transactions, the first non-empty shard proven and paid") | none yet |
| 22 | The base fee is burned in full and the priority fee splits 80% to the miner and provers, 20% to the apps whose code ran | Homepage Economics caption and Build card; litepaper "Where fees go" | tested by the team | repo `f5f8c80`; fork worktree `vendor/igneum-node-exec` | `tools/evm-smoke/smoke.mjs` receipt checks; bench-log "execution layer devnet v3" | Transfer receipt: `burnedProvingFee` 200 gwei, `minerTip` 16,800 gwei (80%), unregistered developer share 4,200 gwei burned; contract call: 80% to the miner, 20% credited to the payee the constructor registered, balance delta equal. 3 October 2026, Apple M5 Max simnet. The provers' part of the 80% is not split out (no provers exist); the base fee stayed at the 1 gwei floor throughout | none yet |
@ -87,6 +87,7 @@ Versions in the table: `igneum-pow` is the Rust crate at `igneum-pow/Cargo.toml`
|---|---|---|---|
| 15 | implemented | implemented, with a live result | the first non-empty shard (block 72704, 29 transfers) proven, verified and paid on the devnet; not every block is proven yet |
| 21 | designed | tested by the team | 388 shards paid from the pool on the live devnet, the rule in `proving.rs`, the numbers in the bench log |
| 22 | Card lifetime: a 4 GB card mines about four years and an 8 GB card about twelve, under the dataset's step schedule (2 GB at genesis, doubling at years 4, 12, 28, 60) with the cache freed after the daily build | Litepaper Hardware and vs RandomX ("Dataset" row); homepage Mine card and "Memory" row | designed | `docs/analysis/card-lifetime-2026-10-05.md` (branch card-lifetime 1fecfe2); spec 1.13.3 option (b) recommended to Josh 5 October 2026 (`docs/plans/counter-asic-2-rollout.md` 6c) | The per-tier working-set arithmetic of that document (GTX 1650, RTX 3050, RTX 3060, RTX 4090 tiers) against the step schedule | A design claim: under the continuous mapping (a) a 4 GB card is out within 1 to 1.5 years and an 8 GB card at 6 to 7.5 years, so the sentence is true only under the step schedule (b), which the spec has not yet fixed (O-1.13) | none yet |
## What would move a row

View file

@ -0,0 +1,75 @@
# Consequences ledger, 5 to 6 October 2026 (night)
The standing consequences reviewer (CLAUDE.md, "Every number carries its consequences"). One row per number whose consequence for a user tier, a chip builder or a public claim nobody had stated or acted on. Tiers: a home miner with one 8, 12, 16, 24 or 32 GB card; a rig; a pool user; Windows, Linux, macOS; NVIDIA, AMD, Apple. Times UTC. State: open, sent (the owner has the message), in work, closed, decision (in `consequences-decisions.md`).
Rows already handled before this ledger opened, for the shape: 15.6 GB mine-and-prove peak (12 and 16 GB profiles, the sweep `memsweep-pc2-pv1`); 9070 XT 18 MH/s (the read-width experiment); the cache never grows (layer 6 option C); aggregation 9.7 s a block (the aggregation-cost agent).
## Round 1 (21:30 to 22:30)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C1 | Fee switch H = 210,000, reached about 18:45Z on 6 October (DAA 137,041 at 22:32:54Z, 1.002 DAA/s averaged since the 15:40Z read; the plans first said 19:50 from 0.965 blocks/s) | `docs/plans/fee-switch-devnet.md` sections 3 and 7; `vendor/igneum-node-0310/igneum/exec/src/rpc.rs` 849 (no `daaScore`, no `feesV1ActivationDaa` in `igneum_exportSegments`); the app's exporter call without `--fees-v1-activation-daa` (`app/igneum-app/src/prover.rs` 516 to 519) | every prover on the devnet (PC 2, the Mac, any 0.3.10 machine) | From the first chain block at or above H the 0.3.10 node cuts 30,000-pgas shards and meters with the v1 table, but the app's exporter sees a dump without the switch and cuts at 7.5 M with the prototype table: the host refuses the fixture (wrong `S_p`) or the node vetoes the statement. Proving on the devnet goes dark at H and nothing is paid until every prover runs a node whose export carries the switch (the proving-v1 fork does, `vendor/igneum-node-pv1` rpc.rs 961 and 982). `pc2-chain.ps1` fixtures from the live node fail the same way after H | Either 0.3.11 (proving-v1 fork) on every prover before H, or the switch republished at a later H before 19:50Z tomorrow (a digest flip, every node). Recommendation in `consequences-decisions.md` D1 | proving v1 acd4f36bc2c07a4e2, shipper ae892a8b0f78fe31c, coordinator ada8afb62d752b1e2 | in work (proving agent: fork eb32c645 on 21d4c73c carries daaScore and feesV1ActivationDaa, the exporter needs no flag; 0.3.11 on every prover before 16:00Z on 6 October or H moves to tip + 86,400; the line is in the proving plan, the rollout plan and release-0.3.10.md) |
| C2 | 16,751 MiB GPU peak during the chain of 8 with the miner resident (`chain-pc2-pv1c`) | bench-log "proving v1", step 2 chain row | 16 GB cards (RTX 5080, 5060 Ti 16 GB, 4060 Ti 16 GB) | The plan's "a 16 GB card sits 0.4 GB under tonight's peak" is the empty-shard row (15,590 MiB). The chained aggregation adds 1.2 GB and lands at 16.4 GB, over a 16 GB card. So a 16 GB card cannot mine and aggregate on this build; it can at most mine and prove shards, with 0.4 GB spare and no full shard measured | The 16 GB chained-aggregation row goes into the sweep; the aggregator step in `prover.rs aggregate_once` gates on card memory (24 GB with the miner running, else pause the miner on that card for the aggregation); the Proving tile says which role the card runs | proving v1 | closed as measured (proving agent, spcurve-miner-pc2-pv1 219517f: the adopted shard beside the miner 22,210 MiB and 13.2 s, so a 24 GB card has 2.3 GB spare from the fee switch, the number approximate for the card itself because it is the 5090's allocation pattern; the prototype shard 30.1 GB, the 32 GB card alone; aggregation_card gates on the same memory rule, 23,552 MB either role). Open: the first real 24 GB card measurement, and the public line says "24 GB" from a 32 GB card's pattern (D2 wording) |
| C3 | 13,816 MiB GPU peak, prover alone, empty shards (`prover-cost-pc2-pv1`) | bench-log "proving v1", step 1 | 12 GB cards (RTX 3060 12 GB, 4070, 5070), the default-on rule (`provedefault.rs` `MIN_VRAM_MB` 11,776) | On the only measurement the prover by itself exceeds a 12 GB card by 1.5 GB, so the 12 GB default gate switches proving on for cards that cannot run it on this build unless the SP1 knobs bring the peak down. The litepaper's "12 GB or more proves full shards" (`site/litepaper.html` 560) and evidence row 16 rest on the sweep | 0.3.11 does not ship the default-on until the sweep has a row under 11.5 GB for a full v1 shard; if none, the gate moves to 24 GB and the public claim is qualified (D2) | proving v1 (sweep in work); public claim: decision | closed as measured (proving agent: the sweep moves no floor, 13.9 GB for an empty shard, 28.3 GB for a full prototype shard; the default is 20 GB mining / 16 GB prove-only; the 12 GB sentence is false on this build and goes to Josh as D2 with the curve); the v1-shard row is C15 |
| C4 | Host RAM 25,550 MB used of 63,132 on PC 2 with the prover on; the WSL2 VM working set 7,915 MB | bench-log "proving v1", step 1 host RAM row | Windows home miners with 16 GB RAM (the common gaming PC); Macs with 16 GB switching the CPU prover on | The prover default has no RAM floor. On Windows the WSL2 VM alone holds 7.9 GB beside the app, the node and the game-class desktop; a 16 GB machine with proving on by default swaps or kills the node. The app's own RSS (engine, node, verifier) is not separated in the measurement, so no requirement can be stated yet | The sweep job records the app's and the node's working sets beside the VM's; `provedefault` reads total RAM and stays off under 32 GB on Windows until measured; the Proving tile and the miner page state the RAM requirement | proving v1; miner UI adbf58b058186a18b (the tile line) | closed in code (proving v1 c2544be: MIN_RAM_MB_WINDOWS 31,000 in provedefault.rs, the tile line names the 7.9 GB VM on a 63 GB PC; unknown RAM is not a gate; the miner UI help line next cut) |
| C5 | The app quit for the 0.3.10 update at 20:01:09Z and aborted the chain job at block 3 ("aborted (the app is quitting)"); `prover.rs` kills the child on quit | bench-log "proving v1" chain run 1; `app/igneum-app/src/prover.rs` 266 to 272 | every prover on every update; with proving v1 the aggregator | Tonight's `update-now` to every app aborts whichever shard each prover has in flight (up to 37 s of work each, no payout, re-assigned to nobody until the window passes). Under proving v1 a restart mid-segment loses the aggregator's chain state: the segment goes unproven after T (600 DAA), the aggregator share of 8 blocks is forfeited to the escrow, and the next record must be fresh-chain. With one aggregator on the devnet every app update costs 10 minutes of unproven segments | The update's safe moment waits for the prover's current proof (as it does for the node); the aggregator persists the last segment proof and resumes the chain after a restart; the Updates section says "waits for the proof in flight". For 0.3.10 tonight the aborted shards are an accepted cost, recorded | proving v1; shipper (tonight's rollout note); miner UI (Updates wording) | refused for tonight by the proving agent (persisting the segment proof and holding the update for a proof in flight are 0.3.12; the cost is in the plan as the rule working as written); the shipper carries the aborted-shard count in release-0.3.10.md section 8; the 20:01:09Z abort was NOT 0.3.10 (shipper: nothing published), see C16. The count, read from the intake at 22:3xZ: PC 2 quit for the 0.3.10 restart at 21:49:29Z (run win-1ccfe586-20261005-200114) with no proof in flight, because its prover had been dark on the root socket since 21:25Z (last exporter line 21:44:02Z, C17); the Mac proves nothing by default; so the restart aborted 0 shards tonight and the first instance of this class will be the 0.3.11 rollout |
| C6 | The prover costs a mining 5090 4.0% (124.7 to 119.7 MH/s); the software dev fee is 1 template in 100 | bench-log "proving v1" step 1; `packaging/hive/README.md` "The dev fee" | pool users; the pool's ledger | A member who proves sends 4% fewer shares, so the pool's vardiff and `stats.hashrate` read a 4% loss while the proving income (90% of the shard credit plus a tenth to an aggregator) is paid to the member's own key and never appears in the pool's ledger: the pool dashboard understates a proving member's earnings. The software dev fee has no mechanism in pool mode (the pool issues the templates), so a pooled miner pays no dev fee today and the pool design must say whether it takes one (1 share in 100 to the dev address) or none | The pool-v0 design states both: the member `stats` carry a `proving` flag and the pool page shows proving income beside shares; the dev fee rule in pool mode is written down before the first pool ships | pool a4781ba117326091f | closed on pool-v0 (pool agent: the member stats line carries a proving flag, /api/miners/<address> and the pool page show it with the 4% note and that proving income never passes through the pool; the software dev fee is NONE in pool mode, the pool's own fee (default 1%) is the only fee, carried in the welcome message's share_scheme, shown on the Connect card, written in docs/plans/pool.md and packaging/hive/README.md). Merge note: packaging/hive/README.md is now edited on three branches (hive-words, ota-k2, pool-v0). Public wording: the miner page's "a visible 1% software fee you can switch off" is a solo-mining sentence once a pool exists, added to D8 |
| C7 | The HiveOS package holds `igneumd`, `igneum-miner` and the two workers; no `igneum-prove-host`, no SP1 GPU server, no per-card rule | `packaging/hive/make-hive-package.sh` 26 to 30; README "What the hooks do" | rigs (HiveOS, Linux), the largest hashrate tier | A rig cannot prove at all, so the 20% proving share is reachable only from the app. The miner page says "the card mines and proves" (`site/miner.html` 7, `site/index.html` 484) and the HiveOS README does not say rigs mine only. A rig that could prove needs the 251 MB SP1 GPU server, CUDA 12.8 and the per-card profile of C2 and C3, and a rig's RAM (4 to 8 GB on most Hive images, approximate) is below C4's floor | The rig installer either carries the prover with the per-card rule and a RAM check, or its README and the miners page say rigs mine only and provers are app machines; `IDENTITIES=8 for a big card, 2 for a small one` gets a threshold in GB | rig installer a3e7b2b03222f5cff | closed (rig installer 88f31d9: gates 23,552 MB for both roles from provedefault.rs 440fd59, the README table per tier; HiveOS hive-words 2d056e8: the same table; open only the miners page sentence, D8) |
| C8 | Dataset 2 GiB at genesis plus 0.5 GiB a year; the working-set rule "under 6 GB on an 8 GB card"; the cache 256 MiB doubling at years 4 and 12; scratch up to 128 KiB per resident warp | spec 01 section 1.13.3; coordinator's budget rule; layer 6 option C; `docs/plans/hot-table.md` section 3 | 4 GB and 8 GB cards; the litepaper's claim | `site/index.html` 461 says "Any 4 GB card" and the litepaper (560) says a 4 GB card mines for about four years and an 8 GB card for more than a decade. At 75% of the card the 4 GB card's dataset room is about 2.4 GiB: under one year. The 8 GB card's room is about 5 GiB after cache, hot table, scratch and buffers: about six years, five and a half with the year-4 cache step. Neither public sentence holds under the schedule | A card-lifetime table per tier (sub-agent, `docs/analysis/card-lifetime-2026-10-05.md`); the public wording is a claim for Josh (D3) | sub-agent (table); decision (wording) | table landed (sub-agent, docs/analysis/card-lifetime-2026-10-05.md, 1fecfe2, merged into ca2-coord at 22:50): 4 GB 1.0 to 1.5 years under (a) and year 4 under (b); 8 GB 6.3 to 7.5 or 12; 12 GB 12 to 13.5 or 12 to 28 depending on whether the cache stays resident; Apple 8 GB 2.6 to 3.4 or 4. The coordinator decided the cache is freed after the daily build and recommends (b) to Josh; the public sentences hold only under (b): D3 and D4 revised |
| C9 | Index mapping option (a) multiply-shift (fades cards) against (b) power-of-two steps 2, 4, 8 GiB | spec 01 section 1.13.3, gate 1 | 8 GB and 12 GB cards | Option (b)'s 8 GiB step (about year 12) ends 8 GB and 12 GB cards on the same day; option (a) fades them one year at a time. The choice is a genesis parameter and a tier consequence nobody has put beside the options | Decision request D4 with the lifetime table of C8 | decision | decision (D4, with the card-lifetime table; the public copy now reads the step schedule as the gate 1 proposal, C31) |
| C10 | The on-die-cache recompute chip's gain is 2.4x at every scratch share under the 6 GB cap; the mixer multiplier x2 brings it to 1.2x, x4 to 0.6x, inside the 10 ms verify gate | `docs/analysis/scratch-soundness.md` finding 2 and section 10; M16 analysis table | chip builders; the site's "under 2x" claim; pool share verification | Layer 3 does not deliver the headline and the rollout plan's own rule (6a) says the public claim is qualified when no share gets the chip under 2x. The lever that does is M16's mixer multiplier, which doubles the CPU verify per warp (0.441 ms to about 0.9 ms at x2): a pool core verifies about 11,000 members' shares instead of 22,000, and node block verification doubles. The corrected SRAM cost ($19 to $37 of silicon per mirror die) means the mirror never stops a funded chip; the cache schedule keeps it above GPU L2 only | The coordinator either carries the mixer x2 into class v3 for the devnet (it is a lottery-hash change like the others; the verify cost per warp is measured with it) or marks the site's "under 2x" as qualified in the public copy level 3 and D5 asks Josh which | coordinator ada8afb62d752b1e2 | closed as a consequence (coordinator: the public claim was qualified at 21:50; at 22:00 the M16 mixer x4 was decided into class v3 behind the same activation; the pool-core and node verification consequences are C14) |
| C11 | RX 9070 XT 17.73 MH/s at 198.9 W (0.089 MH/W) against the RTX 5090 122.30 MH/s at 307.6 W (0.398 MH/W); every read width costs the 9070 XT the same 2.4 G line fetches a second | bench-log AMD telemetry entry; readwidth probe ceilings (status 20:35) | AMD home miners; the "three vendors" copy | Under the "widest latency-bound read" rule w16 moves bytes per hash, not loads per hash, so the AMD card keeps about a seventh of the 5090's hash rate and pays 4.5x the electricity per hash; at a UK tariff of 25 p/kWh (approximate) the 9070 XT spends 4.5x the 5090's pence per IGN. The readwidth decision table carries MH/s only; no MH/W or MH per pound per card per class, and the public copy implies vendor parity | The readwidth table adds MH/W (and MH per pound at list prices, approximate) per card per class; the v3 decision names the AMD consequence; whether the class should favour fewer loads per hash for AMD's 64-byte lines is a consensus choice for Josh (D6) | read-width a451c9935bfb1bc19; coordinator; decision | closed (read-width e752fc7 section 4.1: MH/W and MH per pound per class per card; the AMD gap is the card's random-access rate, no width closes it; the coordinator's 22:30 entry says "near parity per pound" where the table says the 5090 is 2.2x per pound: C18) |
| C12 | The v3 working set on Apple: 1,568 to 1,760 MiB today, 2,592 to 2,784 MiB at the 2 GiB genesis dataset, in unified memory shared with macOS; 2,048 launched warps x 128 KiB scratch | `docs/plans/hot-table.md` section 3 M5 Max row; era-layout section 3 | Apple silicon with 8 GB and 16 GB (the base Mac mini, MacBook Air) | An 8 GB Mac holds the miner's 2.6 GB beside macOS's 3 to 4 GB: it mines today and swaps at the first dataset growth step. The app has no gate on Metal's `recommendedMaxWorkingSetSize`; the site's "Any 4 GB card" has no Apple line. The CPU prover's RAM on a 16 GB Mac is unmeasured (the Settings switch lets it on) | The app reads `recommendedMaxWorkingSetSize` and refuses to mine (with the reason on the Mine tile) when the working set exceeds it; the site says "Mac: 8 GB or more"; the Apple scratch row at 128 KiB is measured in the readwidth table, not launched at 2,048 by assumption | miner UI adbf58b058186a18b; read-width (the Apple row) | in work (read-width 30ff674: the Apple scratch footprint by arithmetic is 1.4 GiB at 2,048 x 32 KiB and 1.6 GiB at 2,048 x 128 KiB, 2.4 GiB at 4,096 x 128 KiB; the Metal harness now prints currentAllocatedSize and recommendedMaxWorkingSetSize in its RESULT line, the measured row waits for the Mac measure lock, held by a prover measurement since 20:31Z; the miner UI gates on the arithmetic plus its output buffer until then, mining.gate_reason next cut) |
| C13 | `/api/supply` `max_supply_ign` 4,000,000,000 against `minted_at_end_ign` about 3,963,000,000 (the ramp withholds about 37 M, the floors the rest); the live tables lack `number`, `tx_count`, `detail` until the observer restarts | `site/api/supply.mjs` 31 to 48; `docs/plans/explorer.md` "Open" | everyone who reads the explorer beside the homepage tile "4B IGN hard cap, ever" (`site/index.html` 386) | The tile says 4 billion and the explorer page says "of 4,000,000,000 by the rule" while the API's own end figure is 3.963 billion: a reader who adds the two columns finds 37 million missing. The explorer page shows "Circulating 0 IGN" on the devnet deployment until the observer restarts on the new code | The tile, the litepaper's supply line and the explorer carry "cap 4,000,000,000; about 3.96 billion ever minted" (the litepaper already says "approached and never reached"); the explorer does not go live before the observer restart lands the columns | explorer a76f60b415859b7b5; wording: Josh (D7, with D3) | closed (explorer 3e01212: the tile reads "cap 4,000,000,000, 3.96 billion ever minted, x% so far" with the withheld figure on hover; the homepage tile and litepaper wording stay with Josh, D3 and D7; the zero-circulating premise was sharper than stated, a 42703 failure not a null, now a 503 with the reason, and moot: the live observer restarted on the explorer code at 19:42:50Z, so the live tables carry the columns) |
| C14 | The M16 mixer x4 decided into class v3 at 22:00 (coordinator): verifier 1.6 to 4.8 ms per warp against 0.441 ms today (0.87 cold), measurement running on branch ca2-mixer | coordinator's reply 21:5x; `docs/analysis/m16-recompute-attacker-2026-10-05.md` table; spec 09 section 9.8 item 5 | pool operators; every node (the Hetzner seeds, the observer, a 2019-class laptop); the 10 ms verify gate | At x4 one pool core verifies 200 to 600 shares a second instead of 2,270, so one core covers 2,000 to 6,000 members at one share per 10 s instead of 22,000; a block's CPU re-check and the miner's own `cpu re-check` of every found hash cost 4 to 11x; the seeds' small VMs verify every header at that cost; the gate's margin falls from 23x to 2 to 6x, which bounds Counter ASIC 3.0's room | The coordinator is adding the numbers to the ca2-mixer document and the status (said in reply); the reviewer checks at the next sweep that the pool-members-per-core and the seed-VM header-verify rows are there, and that the spec 09 figure 2,270 shares a second per core is re-cut with v3 | coordinator ada8afb62d752b1e2 | closed with C19 (the same measurement: x4 is 1.45x and x8 2.1x the verifier, not 4 to 11x; a pool core verifies 1,140 shares a second at x4 and 790 at x8 quiet) |
| C15 | A full prototype shard peaks at 28.3 GB on sp1-gpu-server 6.8.1 (the sweep, proving agent 22:0x); an empty one 13.9 GB; no knob moves either floor | proving agent's reply; sweep job `memsweep-pc2-pv1` | 24 GB cards (4090, 7900-class if it had a path); the 5090 that mines and proves; the fleet table | A 24 GB card cannot prove a prototype full shard at all, mining or not; a 5090 mining (3.4 GB resident) plus a full shard is 31.7 GB against 31.8 GB, the edge. The 20 GB / 16 GB default rests on the empty-shard number. From H tomorrow the fleet proves v1 shards of 30,000 pgas (about 7 M cycles, a ninth of the prototype shard) whose peak is unmeasured; `proving/fixtures/fees-v1-shards2` and `-shards3` are that shape. The fleet table's "proving-only" rows are 32 GB-card rows until then; the litepaper's "12 GB" (D2) and the evidence row 16 fall with it | The sweep's last row is a v1-budget shard with and without the miner, and the provedefault gates are set from it before 0.3.11 ships default-on; the fleet table labels its rows by the card that fits | proving v1 | closed as measured (proving agent, the S_p curve, app default 440fd59): 32 GB mines and proves today (28.3 GB alone, 30.0 beside the miner); 24 GB proves the adopted 30,000-pgas shard (20.4 GB alone, about 22 GB beside the miner) from DAA 210,000 and nothing before it; 16 GB proves only empty shards alone (13.9 GB), nothing beside the miner (15.7 GB); 12 GB proves nothing on SP1 6.8.1; the default is on at 24 GB or more. Relayed to the rig installer and the HiveOS words sub-agent for their tables; until H tomorrow every prover on the devnet is a 32 GB card |
| C16 | PC 2's app quit at 20:01:09Z ("aborted (the app is quitting)"), logged by the proving agent as "the app quit for the 0.3.10 update"; the shipper says nothing of 0.3.10 was published and no update-now of its exists (the live manifest is 0.3.9 from 17:59:14Z) | bench-log "proving v1" chain run 1; the shipper's reply 22:0x | every measurement on PC 2 that straddles 20:01Z (the prover-cost phase B ended 19:51Z, the chain re-run began 20:05Z, the readwidth 5090 job queued) | An app that quits for an unknown reason voids any number taken across it (CLAUDE.md: a number taken while another build or simulation ran is not a number; the same for a restart). The bench-log line names a cause that did not happen | The proving agent reads PC 2's app log for the quit reason at 20:01Z and corrects the bench-log line; if the cause is another agent's job or the auto-update, that agent's measurements across it are marked | proving v1 (the log read); the agent the cause names | closed in part (proving agent: PC 2 logged "quit: stopping the miners, then the node" at 20:01:09Z, a plain quit command 20 s after the efficiency sweep's administrator prompt was cancelled at the keyboard and 13 s after the live prover failed on a root-owned /tmp/sp1-cuda-0.sock left by the chain job; the bench-log line corrected; the quit's origin is not in the log, see C17) |
| C17 | The chain job ran igneum-prove-host as root inside WSL2 and left a root-owned `/tmp/sp1-cuda-0.sock`; the live prover (the app's user) then failed with `CudaClientError: Connect(PermissionDenied)` at 20:00:56Z; the app quit at 20:01:09Z on a command whose source the log does not name | proving agent's reply 22:1x; PC 2's app log | every PC 2 measurement that shares the card with the live prover; every operator whose machine takes remote jobs | Two classes, not one bug. (1) Any job script that runs the SP1 server or the host as root in WSL2 breaks the live prover for every later shard until a reboot or a manual unlink; the proving agent fixed its own scripts (kill the server, remove the socket at the end), the class check (CLAUDE.md, 5 October: grep every script with the same shape, add a check that fails when the shape comes back) is not yet written. (2) A quit the log cannot attribute (job, UI, signal, update) voids the measurements around it and nobody can say who stopped a miner; the app logs "quit:" without a source | (1) `tools/ci/` gets a check that fails on any `.ps1` or `.sh` playbook that invokes `igneum-prove-host`, `sp1-gpu-server` or `wsl -u root` without the socket cleanup line, and the proving agent greps tonight's four PC 2 scripts; (2) the engine's quit log line carries its source (job id and playbook name, the UI, a signal, the updater) in the next app cut | proving v1 (1); coordinator for the next-cut list (2) | closed in part (proving v1 c2544be: every pv1 playbook kills the server and unlinks the socket at start and end, tools/ci/prover-socket-check.sh in CI; the publish-time gate is C27 on bash-body-check 6805125; the unattributed quit, the engine logging its source, is on the coordinator's next-cut list) |
## Round 2 (22:30 to 23:30)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C18 | 5090 0.072 against 9070 XT 0.032 MH/s per pound at list prices (2.2x), 4.9x per watt | read-width.md 4.1 | AMD home miners; the public level 3 copy | The coordinator's 22:30 status entry says "near parity per pound"; the table it cites says 2.2x. The level 3 page is written from the status | The status and level 3 carry the table's figure (2.2x per pound, 4.9x per watt, 7.5x in rate); "near parity" is struck | coordinator | closed (coordinator 6b07776: the status, the rollout plan and the public copy carry 2.2x per pound, 4.9x per watt, 7.5x in rate) |
| C19 | Mixer x4 into class v3: CPU verify 1.6 to 4.8 ms per warp against 0.604 today; pruning depth 108,000 DAA s (spec 02) | status 22:00; M16 table; spec 02 line 17 | every node (the three testnet seeds on small Hetzner VMs, the observer, a laptop node); IBD; pools | A new node verifies every header in the pruning window on one core: 108,000 x 4.8 ms = 8.6 min at x4 against 1.1 min today (a 30-s block-time budget at 1 block/s is unaffected: 4.8 ms per block is 0.5% of a core). Every miner's own `cpu re-check` of a found hash and every pool share verification cost the same 8x; the 10 ms gate (ledger M9) keeps 2x of margin at the slow end, which is what Counter ASIC 3.0 has left to spend. On the 2019-class laptop core the evidence table names (rule 3) the figure is unmeasured and may pass 10 ms | The ca2-mixer document carries: ms per warp on the M5 Max core AND a scaled 2019-class figure (marked approximate), the pruning-window IBD minutes per tier, pool shares per core per second, and the gate margin left for 3.0; the seeds' header-verify load is checked in the testnet go checklist | ca2-mixer af345b1e2c541ffbb; coordinator | closed as measured (ca2-mixer 54bbfcc, mixer-x4.md 6.5): one M5 Max core under a load of 5.6, ms per warp v2 1.33, x4 1.94 (1.45x), x8 2.79 (2.1x), worst cold 1.58 / 2.04 / 2.94; quiet-core scaled 0.60 / 0.88 / 1.26 (approximate); 2019-class laptop 1.5 / 2.2 / 3.2 (approximate, unmeasured); shares per core per second quiet 1,660 / 1,140 / 790 (spec 09's 2,270 re-cut to 1,660), a 22,000-member pool needs 1.3 / 1.9 / 2.8 quiet cores; IBD over 108,000 headers quiet 1.1 / 1.6 / 2.3 min (laptop 2.7 / 4.0 / 5.8); gate margin for 3.0 on the loaded core 8.4 / 8.0 / 7.1 ms. The mixer multiplies the ALU part only, so x8 is 2.1x the verifier. Owed before the level 3 page quotes an absolute: one quiet-core run (the ratios are the measurement tonight). Superseded at 22:07 by the fixed crate: v3 (x8) 2.1 ms per warp near-quiet, 3.4x v2 (see C29) |
| C20 | Layer 9: the epoch length as an era parameter, 600 to 7,200 DAA s (10 min to 2 h), base 3,600 | status 22:30 and 23:00 | rigs (HiveOS and the rig installer), Macs, pools, the seed path | Both rig miners run `--exit-on-seed-change` and re-export the pack on exit 42 (h-run.sh 56 to 61, igneum-miner.sh 67 to 78): at a 10-minute epoch every card's miner restarts six times an hour with a pack export each time, and the restart gap is lost hashing; the app's prepare-ahead path does not restart. The Mac fleet's prepare pause (35 s an hour at one epoch an hour, bench-log M11) becomes 3.5 min an hour, 6%. The 10-minute seed VDF (spec 04) equals the shortest epoch, so the seed for epoch n+1 is known only as epoch n starts, which is the compile-ahead window the agent must measure per card (the 5090 compiled in 1,285 ms, the 9070 XT unmeasured). A pool's `job` cadence and the dev-fee counter are unaffected | The epoch-length document carries a per-tier row: rig restart cost per epoch length (and the fix: prepare-ahead in the rig scripts, no exit 42 path), the Mac pause share, the compile-ahead margin per card at 600 DAA s against the VDF; the rig installer removes `--exit-on-seed-change` in favour of the prepare path before layer 9 can draw a short epoch | ca2-epoch a32a3ece66c02417a; rig installer | closed with a correction (ca2-epoch 4300608, epoch-length.md sections 6.3, 7, 9): the rig scripts pass --prepare-packs as well, so a worker with prepare support swaps in place and exit 42 is the fallback on a prepare MISS (the loaded iGPU missed 2 of 10, M11), not a restart every epoch; the risk at 600 s is one restart plus export plus inline compile per missed boundary; the Mac race pause is 6.3% of a 600-s epoch (race default off); the program is known a full epoch ahead at every length (lead and T_epoch fixed, option A); the 9070 XT compile is OWED (no prepared line from gfx1201 in any upload); the rig installer (88f31d9) confirms the rig already takes the prepare path: exit 42 fires only when a worker's ready line lacks "prepare 1", and both shipped workers answer it; documented in its README, no code change |
| C21 | OTA K2: apps embed `OTA_PUBLIC_KEYS = [K1, K2]` and honour a signed `revoked_keys` list; the rig installer verifies the manifest with ONE key (`OTA_PUBLIC_KEY_HEX`, install-rig.sh 168, igneum-update.sh 2) and knows no revocation; the HiveOS package verifies nothing (no manifest, the override reaches it only by republish) | ota-k2 c722579; packaging/linux; packaging/hive | rigs (both packages), the seeds (if they take the manifest) | The day K1 is lost or revoked and the manifest is signed with K2, every rig on the installer refuses the manifest, stops taking overrides, and is isolated at the next height switch; a leaked K1 keeps signing for rigs, because they carry no revocation list. HiveOS rigs get neither keys nor revocation: a republished package is their only path, and nothing checks who published it | The rig installer carries both public keys and the `revoked_keys` rule in the same form as the app (keys.md section 4, step 3), installed and read from the manifest; the HiveOS README states that the package is unsigned and names the sha256 the Flight Sheet URL should be checked against; keys.md lists the rig and HiveOS paths in its table of what trusts K1 | OTA key af2bb75a5436324d0 (keys.md, the shared verifier form); rig installer a3e7b2b03222f5cff | closed for the rig and the docs (rig installer 88f31d9: OTA_PUBLIC_KEYS [K1, K2 slot] embedded, manifest_check mirrors the app's manifest::check with the revoked_keys record at /var/lib/igneum/updates/revoked.json, tested on three throwaway keys; OTA agent 00fcbb5: keys.md table of every path that trusts K1, the HiveOS README unsigned-archive note); open: the wallet (wallet-v1) still trusts K1 alone, listed in keys.md for its owner; merge note: packaging/hive/README.md is edited on both ota-k2 and hive-words |
Sweep 3 (20:46 Mac clock) notes, no new row: the 9070 XT dropped off PC 1's bus at about 20:40 UTC (the second eGPU fault of the day); the coordinator stated the consequences (G1 on the gfx1036 stand-in, the 9070 XT v3 hash-rate and power rows owed, every earlier 9070 XT row stands, the AMD sweep queue item blocked, nobody woken) in its 21:05 and 21:10 entries and the rollout plan 7b. The epoch-length plan (ca2-epoch 4300608) carries its own per-tier table (section 7), including the node tier (one core 100% busy on the VDF at the 600-s floor) and the chain (24% of blocks in difficulty settle at the floor). The proving agent's 440fd59 rewrote the litepaper's two proving-gate sentences and evidence rows 15 and 16 on its branch (D2: Josh approves the draft); the litepaper's card-lifetime sentence is untouched (D3, D4).
| C22 | The SP1 CPU prover peaks at 29.5 to 30.5 GB RSS whatever the shard size and costs 282 s a shard (PC 1, `cpu-prove-pc1-small2`); no zkVM proves on AMD; the analysis concludes "no CPU tier" | `docs/analysis/amd-proving.md` sections 2, 3, 4a (amd-prove f1d7a7d, merged into ca2-coord) | Macs with 16 or 24 GB (the Settings switch turns the CPU prover on); AMD-only Windows and Linux machines (Settings can switch it on); every tier's expectation of the 20% share | provedefault.rs has a RAM gate for Windows (31,000 MB) and none for macOS or Linux, so a 16 GB Mac that flips the switch runs a 30 GB prover into swap and takes the node down with it (the Mac went down at 1% battery on 4 October; this is the same class of outage from memory). The analysis says the tile must say "about five minutes, paid only when no card proves first" but not that the switch is refused under 32 GB. The public tiers: AMD and Apple miners never see the 20% share on this build (now on the site, 1c8439f) | The CPU-prover switch is refused with the reason on every OS under 32 GB of RAM (the Windows constant generalised: macOS reads hw.memsize, Linux /proc/meminfo), and the tile line carries the 5-minute and 30 GB figures | proving v1 acd4f36bc2c07a4e2 | taken in full (proving agent: the prover loop refuses the CPU path on every OS under 32 GB, Settings cannot bypass it, with the line naming the machine's RAM; the tile line on CPU machines carries the 5-minute, 30 GB, paid-only-if-no-card figures; lands in the next app commit after the gate-test build; the "measured on a 32 GB card, not yet on a 24 GB card" marker is in evidence row 16, both litepaper sentences and the plan) |
Sweep 5 (21:08 Mac clock) notes: the coordinator applied the public proving line on ca2-coord (1c8439f: index, litepaper, miner page: "an NVIDIA card with 24 GB or more proves; AMD and Apple cards mine; a prover for them lands when a zkVM ships one") and the proving agent rewrote the litepaper's two gate sentences on proving-v1 (440fd59): two drafts of overlapping public sentences on two unpushed branches, both for Josh (D2, D8); the integrator takes one. The measure-lock convoy (a0c3d13: a dead holder, two waiters, cargo tests re-acquiring build slots) is stated by the coordinator with a next-cut task. The S_p CPU shard job was dropped on the 312-s small-shard number (stated). GitHub Actions outage: the 0.3.10 installer builds on PC 1 (stated, 21:01).
## Round 3 (21:30 to 22:00 Mac clock)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C23 | Mixer x4 or x8: the 5090's daily dataset build 13.4 ms at x1 (54 ms at x4, about 107 ms at x8); the integrated gfx1036 builds the dataset per PREPARE, 7 to 12 s at x1 and 55 to 124 s under CPU load (epoch-length.md section 7); the x8 rule: "the daily build under 1 s on every card we own" | status 21:25 and 21:27; epoch-length.md section 7 | integrated GPUs (the iGPU tier), 8 GB cards (about a tenth of a 5090's rate), rigs with one weak card | The x8 rule names "every card we own": the gfx1036 is one, and at x8 its per-prepare build is 56 to 96 s per epoch (7 to 16 min under load), so it misses every epoch boundary and falls to the exit-42 path; at x4 it is 28 to 48 s per hour (0.8 to 1.3%) or 4 to 8 min under load. An 8 GB discrete card scales at about a tenth of the 5090: about 1 s at x8, on the rule's edge. The per-day dataset reuse in the worker (owed in epoch-length.md section 9) is what makes the mixer cheap for the iGPU tier; without it the mixer multiplies a per-epoch cost | The mixer decision names the gfx1036 and an 8 GB-class scaled row beside the 5090, M5 Max and 9070 XT in the "under 1 s" check, and the per-day dataset reuse in the worker lands before (or with) class v3, or the iGPU tier is stated as "mines v3 with a restart per epoch" on the level 3 page | ca2-mixer af345b1e2c541ffbb; coordinator | taken (coordinator and ca2-mixer, mixer-x4.md build-time table: the x8 table carries the gfx1036 row and a scaled 8 GB-class row; the under-1-s rule applies to the discrete cards' daily build; for the integrated tier either the per-day dataset reuse in the three workers lands with class v3, asked of the mixer and node agents as a bounded change tonight, or the level 3 page says the iGPU tier mines v3 with a restart per epoch; recorded with the x4/x8 choice) |
| C24 | Two PowerShell job scripts lost a quote inside an inline bash body tonight: amd-prove's awk program (cpu-prove-pc1-small, exit 0 with nothing proved) and the 0.3.10 installer's `bash -c` string (fb-installer-pc1-3, exit 2 in 4 s); amd-prove added `tools/amd-prove/check-job-bash.sh` (bash -n on its own scripts' bash bodies) | amd-proving.md section 2; status 21:28 | every PC job; the morning's rollouts | The class (CLAUDE.md, 5 October: fix the class the same day, add a check that fails when the shape comes back) is "a bash body inside a PowerShell job string"; the check exists for one agent's scripts and did not cover the shipper's, which failed the same way an hour later | A repo-wide CI check: every `.ps1` under relay/playbooks and tools that carries a bash body (`wsl ... bash -c`, `bash -lc`, here-strings fed to bash) has that body extracted and passed through `bash -n`; the shipper's rule (the WSL part as a file run with `bash <file>`) written in packaging/README-ship.md as the convention | sub-agent (bounded, no owner); coordinator told | closed on branch bash-body-check 7adb1ca (tools/ci/bash-body-check.sh with fixtures and self-test, ci.yml, the convention in packaging/README-ship.md; the flagged existing playbooks are in the sub-agent's report for their owners) |
| C25 | Ember Tune: the signed manifest carries per-card-model tuning priors (power limit and core clock) that a new card applies and confirms in two steps; K1 signs it | ember-tune.md sections 4 and 5; bench-log Ember Tune entry | every NVIDIA and AMD card on the app; the keys | The manifest now sets clocks and power limits on every user's card, so the signing key's blast radius grew: a signed prior can underclock the fleet or push a card model to its power ceiling. The plan's clamps (inside power.min_limit / max_limit and clocks.max.gr, the confirm step, a faulted step reverted) are the bound; keys.md's "what K1 signs" table (ota-k2 00fcbb5) predates the priors and does not list them | ember-tune.md section 5 states the bound in one line (a prior can never set a value outside the card's own reported limits, and never a memory clock), with the test that proves it; keys.md's table gains the tuning priors under K1 with that bound | Ember Tune a855dcc4bd05e0615; OTA key agent (the table row) | closed (Ember Tune 5d7ced9: the rule as a row in section 5 with the two tests named, a prior of 9,000 MHz at 30% clamps to 3,090 MHz at 50%, a bad prior costs one confirm step per card; the OTA agent has the keys.md note: tuning priors and the kill switch under K1 with that bound) |
Sweep 7 (21:49 UTC) notes, no new row: the hot table is measured and NOT adopted (g 0.93 to 0.96 on the Mac against the 0.97 rule, bdab8df); the era draw passes the 5% rule on the Mac (spread 0.8%, c570da3) and its chip line says the union of a program's 16 windows covered the whole dataset in 300 of 300 programs, so a chip mirrors the whole dataset or nothing; the mixer x8 passes the verifier half of its rule (C19) and the PC build rows decide the other half about 22:10; PC 2's miners have been off since a job's /api/resume at 21:25 answered ok without restarting them (the devnet short PC 2's rate, the aggregation-cost mining phases void, a next-cut defect: a resume re-checks the miner processes), all stated by the coordinator at 21:45; PC 1 restarted on 0.3.10 at 21:40:41Z with no measurement straddling it.
## Round 4 (22:00 to 22:30 UTC)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C26 | The 13.9 GB floor's cause: the shipped sp1-gpu-server 6.8.1 panics on any card under 24 GB (builder.rs 35 to 39) and allocates its core, recursion, shrink and wrap provers at Setup; the prover-floor agent rebuilds it from source on PC 2 with those sizes cut, CUDA_ARCHS=120 | proving-v1.md 4c82e56; status 22:01 | 12 and 16 GB NVIDIA cards (RTX 3060, 4070, 5070, 5080, 4060 Ti 16 GB): the tier Josh asked for; packaging and signing | A server built for CUDA_ARCHS=120 runs on the 5090 only; the 12 GB tier is sm_86 (3060) and sm_89 (4070), the 16 GB tier sm_89 and sm_120, so a cut-size server measured on the 5090 proves nothing about a 3060 until the on-order 3060 runs it, and the build must list sm_86, sm_89, sm_120 (sm_100 is datacentre) to serve the tier at all. Shipping our own 250 MB CUDA server means the project signs and distributes a build of someone else's prover: it enters the DMG and the WSL2 package, the K1-signed inputs, the SBOM-style notes in evidence.md, and every SP1 upgrade is re-done by hand. Cut buffer sizes do not change the verifying key (prover-side chunking), so no guest re-pin, but the recipe must say so with a verify-segment run on a proof from the rebuilt server | The prover-floor measurement states its arch list and the card it ran on; the 12 GB claim waits for the 3060; the rebuilt server's packaging path (who builds, who signs, where it lands) is a row in the proving plan before 0.3.12, and the public line keeps "24 GB" until the 3060 proves on it | prover-floor agent (through the coordinator); proving v1 | taken (the proving plan carries "A self-built CUDA server (the 12 GB path), before 0.3.12": arch list sm_86 / sm_89 / sm_120 with one measured row per family, the build on PC 1 from a pinned SP1 tag, the Mac signs, placement as wsl2/bin/sp1-gpu-server with its sha256 in payload-inputs.json and the DMG, the evidence.md note, the rebuild at each SP1 upgrade, the gate that verify-segment and verify show the pinned keys unchanged; the 12 GB claim waits for the 3060; the prover-floor agent abefda4c3872f866f has the measurement side) |
| C27 | The root-socket fault recurred at 21:25Z from another agent's job (agg-cost-pc2-1) after the class fix and CI check landed; PC 2's prover was dark 37 minutes; a resume at 21:25 answered ok without restarting the miners | bench-log 1d78979; status 21:45, 22:03 | every PC job; the devnet's proving and hash rate tonight | The class check lives in CI, but PC jobs are published from worktrees by `publish-jobs.sh` and never pass through CI before they run, so a job written on a branch without the check runs the old shape. The check must run where the job is published, not only where the repo is tested | `packaging/ota/publish-jobs.sh add` runs `tools/ci/prover-socket-check.sh` and the bash-body check on the script it publishes and refuses on a failure; the sub-agent on the bash-body check wires both; the resume defect is on the next-cut list (stated) | sub-agent bash-body-check (the wiring); coordinator (the rule) | closed on branch bash-body-check 6805125 (publish-jobs.sh add --kind run runs the bash-body check and prover-socket-check.sh before signing, refuses with the output, never skips; test-publish-jobs.sh 32 passed with four new refusals). Merge notes for the integrator: prover-socket-check.sh exists on both proving-v1 c2544be and this branch (add/add, take the superset here); the socket grep flags tools/amd-prove/pc1-cpu-prove.ps1 (CPU-only, -u root, no GPU server) so that job needs the cleanup lines or an allow-list entry before its next publish, told to the coordinator |
Sweep 8 (22:09 UTC) notes: mixer x8 DECIDED into v3 on the PC rows (the daily 1 GiB build latency-bound on every card: 5090 23 to 25 ms, 9070 XT 72 to 77 ms at every multiplier; the chip row 0.92x with the 3x factor), so the public claim holds with margin (D5 re-cut); the verifier regression (2.2x) bisected to inlining in the mixer's fetch loop and fixed, so the quiet-core figures of C19 return to about 0.6 / 0.9 / 1.3 ms; G6 job 3 failed on a stale fork test (era inside the class), job 4 on the final tree; the integration merge into master has three known conflicts (bench-log append-only, packfile.h and host.c take the ca2-v3 side).
| C28 | One sp1-gpu-server per card on rigs pins about 6 GB of host RAM per server today (under 2 GB with route A's buffers, approximate) | `docs/analysis/proving-methods.md` section 5, rig row (proving-methods e7e0db7) | rigs with several 24 GB or 32 GB cards on the rig installer | A six-card rig that proves on every card pins about 36 GB of host RAM under the shipped server; the rig installer's preflight says 16 GB to prove (a warning, per card not per rig) and its prover unit runs one server, so the moment it moves to one server per card (the analysis's route C) the RAM check is wrong by the card count | The rig preflight scales its RAM warning by the number of proving cards (6 GB each today, the route A figure when measured) and the README's per-card table gains a host RAM column; the one-server-per-card unit lands only with that check | rig installer a3e7b2b03222f5cff | closed (rig installer 086008a: the preflight checks host RAM against 8 GB plus about 6 GB per proving card at the 23,552 MB gate, PROVER_RAM_GB_PER_CARD default 6 marked approximate, the README host RAM row per tier; the one-server-per-card unit lands only with that check) |
Sweep 8a (22:2x UTC) note: `docs/analysis/proving-methods.md` (branch proving-methods e7e0db7, 411 lines) carries its own per-tier table for today, route A, route D and route 3, and a public paragraph; it is the third draft of the proving public line (with proving-v1 440fd59 and ca2-coord 1c8439f), noted under D2 and D8; the route choice is D10.
| C29 | The litepaper's verifier line now reads "Measured 2.1 ms on one loaded Apple M5 Max core for class v3" and the status says "4.8x inside the gate"; the measurement (ca2-mixer 54bbfcc) is 2.79 ms per warp for x8 on the loaded core (2.1 was the RATIO to v2), worst cold unit 2.94 ms, quiet-core about 1.3 ms by scaling | site/litepaper.html on ca2-coord 8e65696; status 22:20 | the public page; every node operator who reads the gate margin | A ratio printed as milliseconds understates the verifier cost by a third and overstates the gate margin (10 / 2.79 = 3.6x, 3.4x on the worst cold unit, not 4.8x). The number is the one a reviewer will re-run first | The line reads "about 2.8 ms per warp on a loaded M5 Max core (about 1.3 ms quiet, approximate), 2.1x the v2 verifier; worst cold unit 2.9 ms; the 10 ms gate leaves 3.4x" until the quiet-core run lands | coordinator ada8afb62d752b1e2 | closed, the reviewer's reading WITHDRAWN in part (coordinator 7e6f77c, status 22:25): the fixed crate's session at 22:07 measured 2.1 ms per warp for class v3 as a MEASUREMENT (worst cold 2.15) on a core at load 5.5, and that binary's v2 figure matched readwidth's quiet 0.61 ms within 1%, so the numbers are near-quiet and the litepaper's 2.1 ms was right; the 2.79 ms I cited was the slow binary's 21:40 session (the inlining regression, since fixed); the gate leaves 4.8x (4.6x on the worst cold unit), 3.4x the v2 verifier. The "about 1.3 ms quiet" scaling is struck everywhere. C19's quiet-core figures are superseded by this session |
| C30 | The 5090 mines in the app at 115.4 MH/s (the power sweep, 22:09 to 22:15Z, hash from the app's API) against 136 to 137 MH/s at device time in every bench row tonight, with the cap not binding (draw 316 W under a 431 W cap, SM at 3,051 MHz) | bench-log e304458; read-width and mixer PC rows | every 5090 owner on the app (and every big card: the gap is the app's job loop, not the kernel) | About 15% of a 5090's hash is lost between the kernel and the app, and the sweep entry explains it away as API sampling. M11 measured the 9070 XT at the app's 2^21 job size equal to its 2^24 rate, but no 5090 row exists at 2^21; the 5090 finishes a 2^21 job in about 15 ms, so per-job launch, read-back and template work can cost that much. The STATUS line prints "wall" and "inside jobs" rates and would show it | One measurement on PC 1: `igneum-worker-cuda --bench` on the live pack at --batch-log2 21 and 24 on the 5090, and the 5090's STATUS wall-against-inside gap over 10 minutes; if the job size is the cause, the app's job size for cards over 100 MH/s rises (2^22 or 2^23) in the next cut: a 15% gain for every 5090 owner | repro-bench agent a0b9f574775ef1693 (its PC 1 slot); coordinator | taken (repro-bench agent: --bench at 2^21 and 2^24 on the 5090 on the genesis pack and the live pack in its PC 1 slot, then PC 2; the STATUS wall-against-inside gap from the 10 minutes before and after its window; the consequence written either way) |
Sweep 9 (22:2x UTC) notes: the devnet at 22:24:31Z reads DAA 136,578, 0.909 blocks/s measured over the stats window and 1.005 DAA/s averaged since the fee-switch plan's 15:40Z read (112,227), so H = 210,000 lands between about 18:50 and 19:35 UTC on 6 October, up to an hour EARLIER than the 19:50Z written in fee-switch-devnet.md, the rollout plan 7a and release-0.3.10.md; the 16:00Z check (D1) keeps about 2.8 hours of margin and stands; the plans' ETA is re-cut in D1 and sent to the coordinator. Gates tonight: G3, G4, G6 green on the final class (x8 + era); G4b added (the Mac app passes --prepare-packs only to non-Metal workers, so every Mac would stop at the first v3 epoch: found by the node agent, the fix with a unit test and a real Metal gate run before the ship); G1 and G2 on the PC 1 job since 22:16.
Merge note for the integrator (22:3x UTC): branch `consequences` is docs/plans/consequences-2026-10-05.md and consequences-decisions.md only (base ca8d9f3); a merge-tree against master 1f0d62c shows 0 conflicts. Branch `bash-body-check` (7adb1ca, 6805125, e3bd761) carries tools/ci/bash-body-check.sh, kit-path-check.sh, the copied prover-socket-check.sh (add/add with proving-v1's: take bash-body-check's), ci.yml steps, the publish-jobs.sh gate and packaging/README-ship.md; it is on the coordinator's ship order after ca2-coord.
| C31 | The copy the ship takes (ca2-coord at 22:3x): litepaper line 562 "12 GB or more proves full shards" beside line 452 "Proving needs an NVIDIA card with 24 GB or more"; evidence row 16 still "A 12 GB card proves one shard in about 20 s, designed" while proving-v1 440fd59 withdrew it; line 562 states the dataset "doubles on a step schedule fixed at genesis (years 4, 12 and 28)" | ca2-coord site/litepaper.html 452 and 562, docs/evidence.md 43; proving-v1 440fd59; D4 | every reader of the litepaper; the integrator; Josh's D4 | One page says 12 GB and 24 GB for the same thing; the evidence table on the ship branch contradicts the measurement, and the two branches will conflict on evidence.md and litepaper.html at the merge (ship order: ca2-coord before proving-v1), so the stale row can win by accident; and the step schedule is written as a genesis fact while the coordinator's own 22:50 entry calls mapping (b) a recommendation for Josh (gate 1, D4): a public page should not decide a genesis parameter before he does | On ca2-coord: strike "12 GB or more proves full shards" from line 562 (line 452 is the sentence); take proving-v1's evidence row 16 (WITHDRAWN, 24 GB measured) at the merge and say so in the merge plan; write the growth sentence as the recommendation it is ("the plan is a step schedule ... decided at gate 1") until D4 is taken | coordinator ada8afb62d752b1e2 | closed on ca2-coord a7be43f and 0d9b23d (the 12 GB clause struck; the merge rule "take proving-v1's row 16 and its proving sentences" in the rollout plan; one wording in all five places, "the proposed schedule, fixed at the testnet genesis: 2 GB, doubling at years 4, 12 and 28", the lifetime sentences kept as consequences of the proposal and marked approximate) |
| C32 | The 0.3.10 install at 21:49Z cleared PC 1's app jobs folder and with it the AMD kit fetched at 21:23:59Z; the amd-card-test playbook now says its fetch must be republished after any app update | bench-log 39f02ff; status 21:05 ("the jobs folder is cleared by fetch jobs" was the earlier, wrong reading) | every PC job tonight and tomorrow; the 0.3.11 rollout | A class, not one playbook: every fetch-then-run pair (era, hot table, mixer, repro, Ember, the AMD sweep, the prover-floor build) loses its kit when an update lands between the fetch and the run, and the run fails in seconds or, worse, runs against a stale copy. The 0.3.11 update-now reaches PC 2 while the prover-floor agent's 60 to 90 minute server build runs there (go at 22:17, to about 23:50): if that build's working directory is under the app's jobs folder, the update wipes it mid-build and the 12 GB rows slip past the morning | (1) The 0.3.11 update-now is sequenced after the prover-floor build closes, or the build's directory is confirmed outside the jobs folder before the ship; (2) every run playbook begins with a presence check of its kit and fails with "kit missing: republish the fetch after the app update" (the class check: the bash-body sub-agent's CI check gains a rule that a run job naming a kit path tests it first, or the coordinator's queue re-fetches after every update as a rule) | coordinator ada8afb62d752b1e2 (the queue and the ship order) | taken (coordinator: the prover-floor build lives under /opt/igneum-floor in WSL2, outside the jobs folder, but it is the app's job process and an app restart ends it, so the 0.3.11 update-now goes to PC 2 only after floor-build-3 closes, the ship's earlier steps not waiting; the re-fetch rule and the presence-check rule are in the rollout plan beside the one-job rule, the playbook owners carry it at their next publish; the CI side is with the bash-body sub-agent as a kit-path check). CI side closed on bash-body-check e3bd761: tools/ci/kit-path-check.sh in ci.yml and in publish-jobs.sh add --kind run (34 tests pass); every existing kit-using playbook (13 across master, ca2-v3, ca2-analysis, rdna4-telemetry) already checks before use, so the gate guards the shape without a backlog |
| C33 | `/api/live` at 22:33Z: 13 of 482 blocks fully proven in 10 minutes (2.7%), 1 prover, median proof lag 46 s, the live node's verifier "Off" with 42 pool entries pending and 0 verified; `/api/stats` (the documented public API) carries no proving field at all | live and stats handlers (`site/api/stats.mjs` FIELDS; the explorer branch 3e01212); the homepage "~60 s to a proof"; evidence row 15 | everyone who reads the public API or the homepage tile; the testnet's first external reader | The public stats API hides the one number that qualifies the tile and row 15: coverage is 2.7% with one prover, and the proof lag is 46 s. A reader can find it only on /api/live. The live node's verifier "Off" (42 pending, 0 verified) is the Mac app node in trust mode or without a host, so the page says "verifier Off" while the chain pays provers: a public-page oddity the morning reader will ask about | `/api/stats` gains a `proving` object from the same live_state (`blocks_10m`, `blocks_fully_proven_10m`, `shards_paid_10m`, `provers_10m`, `median_proof_lag_s`, `active`), documented in docs/api/public-stats.md and in its contract test; the live page's verifier line names which node it reads and why it is off; the homepage tile's "~60 s" caption cites the measured 46 s median and the 2.7% coverage ("the target; today one prover covers 2.7% of blocks at a 46 s median") | explorer a76f60b415859b7b5 (the API); the tile caption: D2 wording for Josh | closed on explorer d7e797c (/api/stats carries the proving object with coverage_10m, 0.0273 at 22:33Z, in the contract test, the live check and docs/api/public-stats.md with the sentence that coverage is what a third party reads before "every block is proven"; the live page's proving legend and the API note say the observer's node runs its own verifier off and reads paid shards from the chain). The homepage tile caption stays with Josh (D2) |
Sweep 11 (22:38 UTC) notes, no new row: gate G4b GREEN (a real Metal miner across a v3 boundary; a second Metal-only fault found and fixed first, 00c55aa: serveDataset keyed the day dataset by day alone and would have hashed v3 over an x1 dataset; the coordinator's next-cut rule: every worker path mines across a boundary in the gate network before a class change ships). The reviewer checked the PCs' side of that class: the CUDA worker and the OpenCL host build each resident pair (program, cache, dataset) from the pack's own memhard.h and key it by (epoch, day, class, era) through pairIsClass (proto-cuda/nvrtc/worker.cpp 451, proto-opencl/host.c 1106 on ca2-v3 fa3c932), so the fault does not reach the PCs at N4. The patched sp1-gpu-server built green on PC 2 at 22:32:29Z with sm_86, sm_89, sm_120 (C26's arch list), the floor sweep (9 points) running; the integration merges (readwidth 30ff674, origin/master 1f0d62c) on ca2-v3 49c7e78 with every check green. All gates but G5 (the ship's build) are green; the ship waits on the merged tip.
| C34 | The rollout plan's packaged override line (counter-asic-2-rollout.md 26) carries five switches: difficulty_v2 33000, proving_v0 84100, fees_v1 210000, finality_v3 135200, program_class_v3 N4; section 8a says 0.3.11 publishes ONE object with both activations set | counter-asic-2-rollout.md 26 and 8a; proving-v1.md (the four v1 fields enter the digest only once proving_v1_activation_daa is set) | every node and every prover on the devnet at the 0.3.11 publish | If the publisher copies line 26, proving v1 ships in the binary and never activates: proving_v1_activation_daa stays at never on every node, the digest is the five-field one, the segment records are never carried, and C1's fix (the export fields) still works but the aggregator, the chain rule and the unproven rule stay off while the plan and the public copy say they are live. The four fields (activation_daa H1 = tip + 14,400 at publish, segment_blocks 8, unproven_daa 600, aggregator_share_bps 1000) must be in the packaged line, the manifest object, the hand nodes' and the seed's files, verbatim, and the expected digest read on a scratch node with all nine fields | Line 26 and the ship step name the nine-field object with N4 and H1 both set at publish and the scratch-node digest read over that object (the 22:xx "expected 0.3.11 digest with the two new fields at never" is the rolling-upgrade digest, not the activation one; both are recorded) | coordinator ada8afb62d752b1e2 (the ship runbook) | sent |

View file

@ -0,0 +1,32 @@
# Consequence decisions for Josh (night of 5 October 2026)
Sibling of `ledger-decisions.md`. Each is a consequence of a measured number that needs Josh: money, a public claim, or a consensus parameter outside tonight's delegation. The row number points at `consequences-2026-10-05.md`. Recommendation first, then the options.
## The eleven in one glance (what Josh does, in the order they bite)
| # | By when | One line | What Josh does |
|---|---|---|---|
| D1 | 16:00 UTC, 6 October | The fee switch at H = 210,000 (about 18:45 UTC by the devnet's DAA rate at 22:33Z, 1.002 DAA/s since 15:40Z; an hour earlier than the plans' 19:50) darkens every 0.3.10 prover; 0.3.11 must be on every prover first, else H moves | Nothing if the morning check passes; the coordinator holds it. Know it exists |
| D9 | this week | No 12 GB card exists here; every 12 GB number is scaled from the 5090 | Confirm whether the 3060 and 5060 Ti 16 GB that evidence.md says are on order are real; if not, buy one 12 GB card (about £250 to £400) |
| D2 | before the next site push | The litepaper's "12 GB proves" is false on this build; three drafts of the replacement exist (proving-v1 440fd59, ca2-coord 1c8439f, proving-methods.md section 5) | Pick one; the proving-methods paragraph is recommended |
| D8 | the same push | "The card mines and proves" and "a visible 1% software fee" are solo-NVIDIA sentences now | Approve the qualified wording |
| D3, D7 | the same push | "Any 4 GB card", "4 GB about four years, 8 GB more than a decade", "4B hard cap" against the lifetime table and the 3.96 billion ever minted | Approve the re-cut sentences |
| D4 | gate 1 (before the testnet genesis) | Dataset growth mapping: (b) power-of-two steps (years 4, 12, 28, 60) keeps the public sentences true; (a) fades cards one year at a time | Choose; (b) recommended with the step calendar published |
| D5 | before the next site push | The chip claim: mixer x8 measured at 0.92x with the 3x factor, "under 2x" holds with margin on the stated convention | Confirm the claim stays, with the convention named |
| D6 | Counter ASIC 3.0 | AMD mines at a seventh of a 5090 and 4.9x the electricity per hash; no read width closes it | Accept for v3; the miners page says so |
| D11 | before the next site push | The hero, the abstract and the level 1 copy now say "the model and the bounty are public" / "bounty standing"; funding.md prices the chip bounty at USD 50,000, marks it NOT FUNDED, and its rule 3 says a bounty is announced only when escrowed | Strike "and the bounty" from the hero and the abstract until the USD 50,000 is escrowed, or escrow it; the litepaper's older "a bounty is attached" (finality, line 686) is the same question |
| D10 | after route A's rows and D9's card | Proving route: route A (re-sized SP1 server) now; RISC Zero as a second proof family (Apple, and the fallback) is a consensus and verifier change | Measure both; adopt a second family only on your say |
| # | Row | What needs deciding | Recommendation | Why |
|---|---|---|---|---|
| D1 | C1 | The fee switch lands at H = 210,000 about 19:50Z on 6 October, and the 0.3.10 node's export does not carry the switch, so every app prover's shards are refused or vetoed from H. Move H, or race 0.3.11 onto every prover first | If 0.3.11 (with the proving-v1 fork, whose export carries `daaScore` and `feesV1ActivationDaa`) is not on every prover by 16:00Z on 6 October, republish the override with H = tip + 86,400 rounded to the next thousand, re-read the digest on a scratch node, every node in one sweep (the fee-switch plan's own rule for a later H). The coordinator holds the devnet delegation; this note is so the morning does not find proving dark | A dark proving pool on the devnet costs nothing on chain (the escrow keeps it) but every coverage, latency and fleet number measured after H is void |
| D2 | C3 | The litepaper says "12 GB or more proves full shards"; the only measurement puts the prover alone at 13.8 GB on a 32 GB card | The sweep and the S_p curve are in (proving agent, 5 October late): no knob moves the floor; 12 GB proves nothing on SP1 6.8.1, 16 GB proves only empty shards alone, 24 GB proves the adopted 30,000-pgas shard (20.4 GB alone, about 22 GB beside the miner), 32 GB proves everything. Change the sentence to "24 GB or more proves; 32 GB proves and mines on one card" and evidence row 16 to tested-by-the-team on those rows; the 12 GB figure returns only if a smaller GPU server or a smaller shard measures under 12 GB. The proving agent has drafted the two replacement sentences on its branch (app 440fd59, litepaper and evidence rows 15 and 16); nothing is pushed, so Josh approves or rewrites the draft rather than starting from the measurement. Two drafts exist: the proving agent's litepaper sentences (proving-v1 440fd59) and the coordinator's site, litepaper and miner-page line (ca2-coord 1c8439f); the integrator keeps one. One marker is owed on either: the 24 GB figure is the 5090's allocation pattern on a 32 GB card (22.2 GB beside the miner), not a measurement on a 24 GB card, and evidence rule 3 wants that said until a 4090 or a 5080-class 24 GB card runs it | A public number that the first 3060 owner disproves is the FUD the ledger exists to prevent |
| D3 | C8, C13 | The homepage says "Any 4 GB card" and the litepaper says a 4 GB card mines for about four years and an 8 GB card for more than a decade; the tile says "4B hard cap" while the rule mints about 3.96 billion | Replace with the lifetime table's numbers once the sub-agent lands it: "4 GB cards mine at launch; 8 GB for about six years; 12 GB for about fourteen; the dataset grows half a gigabyte a year" and "cap 4 billion, about 3.96 billion ever minted" | The schedule is public and the arithmetic is one line; a reader will do it |
| D4 | C9, C8 | Index mapping at gate 1, now with the lifetime table (`docs/analysis/card-lifetime-2026-10-05.md`): (a) multiply-shift, continuous growth: 4 GB cards out at 1.0 to 1.5 years, 8 GB at 6.3 to 7.5, 12 GB at 12 to 13.5, Apple 8 GB at 2.6 to 3.4; (b) power-of-two steps at years 4, 12, 28, 60: 4 GB to year 4, 8 GB to year 12, 12 GB to year 28 (the cache freed after the build, decided by the coordinator at 22:50), each tier ending on a step day | (b), with the step calendar published on the miners page from day one (the years 4, 12, 28 and 60 named beside the tiers), and the 1 GiB vectors kept. Revised from (a) at 23:0x: the table shows (b) is the only mapping under which the litepaper's "4 GB about four years, 8 GB more than a decade" holds, and a step day known twelve years ahead is a schedule, not an event; the coordinator recommends the same | Under (a) both public sentences are wrong today by 2.5 to 5 years; under (b) they hold and the cliffs are dated |
| D5 | C10 | The site's "under 2x" chip claim: the scratch layer leaves the on-die-cache recompute chip at 2.4x at every share; the mixer multiplier x2 brings it to 1.2x at twice the CPU verify cost (about 0.9 ms per warp, half the pool members per core) | Overtaken by the coordinator's delegated decisions (21:50 to 21:28 Mac clock): the public claim was qualified, the mixer x4 is in class v3 (chip row 0.61x bare, 1.84x with a 3x fixed-function factor, 1.53x at equal silicon with the mirror deducted: "under 2x" holds on the equal-silicon convention and is thin), and x8 is built and measured beside it (0.92x with the factor, from the M16 table); x8 goes in if the verify stays under 10 ms and the daily build under 1 s on every card we own (C23 asks that the iGPU tier be named in that rule). Measured at 21:40 UTC (ca2-mixer 54bbfcc): x8 is 2.1x the v2 verifier (2.79 ms per warp on a loaded core, 1.26 quiet by scaling), 7.1 ms of the 10 ms gate left, the Mac 1 GiB build flat at 21 ms; the verify half of the x8 rule passes, the PC build rows are owed. Then at 22:06 UTC x8 was decided into v3 on the PC build rows (latency-bound on every card): the chip row reads 0.92x with the 3x fixed-function factor, so "under 2x" holds with margin and the level 3 page can state the convention and the margin. For Josh: confirm "under 2x" stays, now with the measured row behind it | The claim is the project's first public sentence on chips; it is either true by a measured lever or it is marked |
| D6 | C11 | Whether class v3 should favour AMD (fewer, wider loads per hash) at a cost to the 5090's latency-bound share, or accept that AMD cards mine at about a seventh of a 5090 and 4.5x the electricity per hash | Accept it for v3 and say so on the miners page ("NVIDIA first; AMD mines at a lower rate per watt on this class"); open the AMD question as a Counter ASIC 3.0 item with its own measurement | Tonight's rule was Josh's and the 5090 margin is the anti-chip argument; AMD's position is a public-copy question, not a gate |
| D7 | C13 | Same as D3's supply wording | with D3 | |
| D8 | C7 | The miner page says "One click: install, start, the card mines and proves" (site/miner.html 7, 13, 21; site/index.html 484). On HiveOS and the rig installer a rig mines only until a Linux prover build is published, and on the app a 12 GB card mines only, a 16 GB card proves with the miner paused, 20 GB and up does both (the sweep of 5 October, before the v1-shard row) | Qualify the sentence on the miner page and the homepage card: "the card mines; 24 GB cards prove too, 32 GB does both at once; rigs mine until the Linux prover ships". The same page's "a visible 1% software fee you can switch off" becomes "solo mining carries a 1% software fee you can switch off; in a pool the pool's own fee is the only one" once pool-v0 ships (the pool agent fixed the rule: no software dev fee in pool mode) | The sentence is the product's first promise and tonight's measurement bounds it by card |
| D9 | C3, C15, C26 | No 12 GB NVIDIA card exists in the fleet, so every 12 GB number tonight is scaled from the 5090 and labelled approximate. The 13.9 GB floor is SP1's GPU server code (`docs/analysis/proving-methods.md`, branch proving-methods e7e0db7, section 1.3: builder.rs adds 4 GB to the card's physical memory and panics under 24, so a 16 GB card (16 + 4 = 20) and a 12 GB card never start and 20 GB is the smallest that does; the trace is allocated at the maximum shard; the CUDA mempool never releases; the proving plan's "under 24" (4c82e56) is the same test read before the addition). The prover-floor patch and the per-card profiles cannot be measured without the hardware | Buy one 12 GB NVIDIA card this week for PC 1's spare slot (an RTX 3060 12 GB or 4070 12 GB, about £250 to £400, approximate; the coordinator's request). Check first whether the RTX 3060 and the RTX 5060 Ti 16 GB that evidence.md row 16 says are "on order" are real and arriving; if so, no purchase, only the delivery date. No public line says "12 GB proves" before a real 12 GB card runs the rebuilt server on the S_p-curve fixtures and recipe | Money, and the one measurement every 12 GB claim rests on; the prover-floor agent's rows tonight replace the approximate figures when they land |
| D10 | C3, C15, C26, C28 | The proving route for the 12 GB tier and for Apple: `docs/analysis/proving-methods.md` section 4 ranks (1) route A, a re-sized SP1 GPU server with `S_p` as the dial and one server per card on rigs, nothing in consensus moving; (2) route D, RISC Zero 3.0.x as proof-system version 2 (the only shipped prover with a documented sub-12 GB configuration and a Metal path), the fallback if A misses 11 GB and the Apple route either way, 3 to 4 days plus a 3-month two-verifier overlap and three spec 7.8 rules; (3) a sumcheck family without a codeword (Jolt-class) in years, not now. Route A is already running tonight; route D adds a second proof family to the node, which is a consensus and verifier change outside tonight's delegation | Route A on the prover-floor rows, gated as the document says (the adopted shard under 11 GB alone and under 60 s prove-only on a real 12 GB card, D9); route D's measurement (RISC Zero at po2 19 and 20 on PC 2 and the Mac's Metal row) may run as a measurement, but adopting a second proof family waits for Josh and for route A's result; the public paragraph of section 5 ("Proving runs on NVIDIA cards with 24 GB or more today. A build for 12 GB and 16 GB cards is being measured ...") is the honest line meanwhile and is the one of the three drafts to prefer, because it names what is being measured instead of a tier | A second verifier in the node is the kind of change the testnet's genesis must carry from day one; measuring it costs nothing, adopting it is Josh's |
| D11 | C29 context; `docs/plans/funding.md` 36 and 63 | The chip bounty on the public pages: "the model and the bounty are public" (hero, abstract, ca2-coord cabec3b), "bounty standing" (level 1 copy); funding.md: USD 50,000 standing, "Not funded", "a bounty is announced only when it is escrowed"; the litepaper already says "a bounty is attached" to the finality review (line 686) | Strike "and the bounty" from the hero and the abstract and "standing" from level 1 until the money is escrowed, and say "a bounty follows the external review" if a sentence is wanted; or escrow USD 50,000 (and the USD 25,000 finality bounty) and keep the words. Applied at 22:25 (ca2-coord 7e6f77c): "and the bounty" struck from the hero and the abstract, "standing" from level 1, the copy says "a bounty follows the external review"; the words return only once escrowed | A public promise of money the project has not set aside is the FUD the ledger exists to prevent, by the project's own funding rule |

View file

@ -0,0 +1,381 @@
{
"pass": true,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1f010000",
"template_switch": {
"epoch": 3,
"daa": 180,
"at": 168
},
"run_ended_at_s": 298.3,
"final_daa": 300,
"blocks": {
"total": 305,
"before_boundary": 181,
"after_boundary": 124,
"chain_before": 176,
"chain_after": 123
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "8f8806638d59850f",
"seed": "234e082d653dc69d",
"miners": 3,
"disagree": false,
"ready_ms": [
219,
235,
232
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "fd9562df32a68313",
"seed": "9b36731951ad7fb3",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "1ae6d90ab299154c",
"seed": "ce88239dfe686691",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "5d0dedd9fd9e29a1",
"seed": "f177ab855facbeb4",
"miners": 3,
"disagree": false,
"ready_ms": [
185,
179,
183,
177,
183,
177
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "e81808dcdb02ce05",
"seed": "bcbc22513d3b73a2",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "06aff9c1d33e7a13",
"seed": "8d6755bc0eb15465",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
}
],
"accepted_per_miner": [
97,
103,
104
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"082fd39ba65df2ff",
"082fd39ba65df2ff",
"082fd39ba65df2ff"
],
"block_counts": [
304,
304,
304
],
"tips_per_node": [
1,
1,
1
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.7,
"daa": 0,
"epoch": 0,
"class": 2,
"nodes": [
"0/234e082d",
"0/234e082d",
"0/234e082d"
]
},
{
"t": 19.7,
"daa": 11,
"epoch": 0,
"class": 2,
"nodes": [
"11/c036063c",
"11/c036063c",
"11/c036063c"
]
},
{
"t": 34.7,
"daa": 34,
"epoch": 0,
"class": 2,
"nodes": [
"34/62ae0e8e",
"34/62ae0e8e",
"34/62ae0e8e"
]
},
{
"t": 49.8,
"daa": 53,
"epoch": 0,
"class": 2,
"nodes": [
"53/c21d7e86",
"53/c21d7e86",
"53/c21d7e86"
]
},
{
"t": 64.8,
"daa": 70,
"epoch": 1,
"class": 2,
"nodes": [
"70/9cb940da",
"70/9cb940da",
"70/9cb940da"
]
},
{
"t": 79.8,
"daa": 88,
"epoch": 1,
"class": 2,
"nodes": [
"88/c84b8514",
"88/c84b8514",
"88/c84b8514"
]
},
{
"t": 94.9,
"daa": 103,
"epoch": 1,
"class": 2,
"nodes": [
"103/1e2dfed3",
"103/1e2dfed3",
"103/1e2dfed3"
]
},
{
"t": 109.9,
"daa": 111,
"epoch": 1,
"class": 2,
"nodes": [
"111/aa7ea4cd",
"111/aa7ea4cd",
"111/aa7ea4cd"
]
},
{
"t": 124.9,
"daa": 139,
"epoch": 2,
"class": 2,
"nodes": [
"139/d52be92d",
"139/d52be92d",
"139/d52be92d"
]
},
{
"t": 139.9,
"daa": 153,
"epoch": 2,
"class": 2,
"nodes": [
"153/02a4323f",
"153/02a4323f",
"153/02a4323f"
]
},
{
"t": 155,
"daa": 165,
"epoch": 2,
"class": 2,
"nodes": [
"165/f2e38b3d",
"165/f2e38b3d",
"165/f2e38b3d"
]
},
{
"t": 170,
"daa": 183,
"epoch": 3,
"class": 3,
"nodes": [
"183/98f2ca9d",
"183/98f2ca9d",
"183/98f2ca9d"
]
},
{
"t": 185.1,
"daa": 195,
"epoch": 3,
"class": 3,
"nodes": [
"195/eecbf07e",
"195/eecbf07e",
"195/eecbf07e"
]
},
{
"t": 200.1,
"daa": 205,
"epoch": 3,
"class": 3,
"nodes": [
"205/ad4a2646",
"205/ad4a2646",
"205/ad4a2646"
]
},
{
"t": 215.1,
"daa": 221,
"epoch": 3,
"class": 3,
"nodes": [
"221/f8626ffd",
"221/f8626ffd",
"221/f8626ffd"
]
},
{
"t": 230.2,
"daa": 233,
"epoch": 3,
"class": 3,
"nodes": [
"233/d94e1815",
"233/d94e1815",
"233/d94e1815"
]
},
{
"t": 245.2,
"daa": 251,
"epoch": 4,
"class": 3,
"nodes": [
"251/f724fd9e",
"251/f724fd9e",
"251/f724fd9e"
]
},
{
"t": 260.2,
"daa": 266,
"epoch": 4,
"class": 3,
"nodes": [
"266/b91936fd",
"266/b91936fd",
"266/b91936fd"
]
},
{
"t": 275.2,
"daa": 276,
"epoch": 4,
"class": 3,
"nodes": [
"276/7c8f720d",
"276/7c8f720d",
"276/7c8f720d"
]
},
{
"t": 290.3,
"daa": 297,
"epoch": 4,
"class": 3,
"nodes": [
"297/63eb2574",
"297/63eb2574",
"297/63eb2574"
]
}
]
}

View file

@ -0,0 +1,378 @@
{
"pass": true,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1f010000",
"template_switch": {
"epoch": 3,
"daa": 180,
"at": 171
},
"run_ended_at_s": 292.3,
"final_daa": 300,
"blocks": {
"total": 305,
"before_boundary": 181,
"after_boundary": 124,
"chain_before": 180,
"chain_after": 122
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "8f8806638d59850f",
"seed": "234e082d653dc69d",
"miners": 3,
"disagree": false,
"ready_ms": [
177,
177,
178
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "bb8dd9ddbf9eb63f",
"seed": "2f66af56da44ca92",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "8ee7a9f33d418e48",
"seed": "5194dfa8a53259bc",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "2d278041ba482dba",
"seed": "54c353de1d8609d7",
"miners": 3,
"disagree": false,
"ready_ms": [
181,
179,
179
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "2ae786d294a8a59d",
"seed": "8529c69223a2194c",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "bc36813df2f41b5f",
"seed": "fdc233b69c42ffea",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
}
],
"accepted_per_miner": [
102,
102,
100
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"712c1b212091dcdc",
"712c1b212091dcdc",
"712c1b212091dcdc"
],
"block_counts": [
303,
303,
303
],
"tips_per_node": [
1,
1,
1
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.6,
"daa": 0,
"epoch": 0,
"class": 2,
"nodes": [
"0/234e082d",
"0/234e082d",
"0/234e082d"
]
},
{
"t": 19.7,
"daa": 18,
"epoch": 0,
"class": 2,
"nodes": [
"18/5b5f2d36",
"18/5b5f2d36",
"18/5b5f2d36"
]
},
{
"t": 34.7,
"daa": 51,
"epoch": 0,
"class": 2,
"nodes": [
"51/333c61cc",
"51/333c61cc",
"51/333c61cc"
]
},
{
"t": 49.8,
"daa": 67,
"epoch": 1,
"class": 2,
"nodes": [
"67/f1fcbc43",
"67/f1fcbc43",
"67/f1fcbc43"
]
},
{
"t": 64.8,
"daa": 89,
"epoch": 1,
"class": 2,
"nodes": [
"89/91180594",
"89/91180594",
"89/91180594"
]
},
{
"t": 79.8,
"daa": 101,
"epoch": 1,
"class": 2,
"nodes": [
"101/49dd0422",
"101/49dd0422",
"101/49dd0422"
]
},
{
"t": 94.9,
"daa": 111,
"epoch": 1,
"class": 2,
"nodes": [
"111/a65a10b3",
"111/a65a10b3",
"111/a65a10b3"
]
},
{
"t": 109.9,
"daa": 122,
"epoch": 2,
"class": 2,
"nodes": [
"122/f06b1eb3",
"122/f06b1eb3",
"122/f06b1eb3"
]
},
{
"t": 124.9,
"daa": 140,
"epoch": 2,
"class": 2,
"nodes": [
"140/bbfe6710",
"140/bbfe6710",
"140/bbfe6710"
]
},
{
"t": 140,
"daa": 155,
"epoch": 2,
"class": 2,
"nodes": [
"155/a4e736dd",
"155/a4e736dd",
"155/a4e736dd"
]
},
{
"t": 155,
"daa": 171,
"epoch": 2,
"class": 2,
"nodes": [
"171/c754adeb",
"171/c754adeb",
"171/c754adeb"
]
},
{
"t": 170,
"daa": 179,
"epoch": 2,
"class": 2,
"nodes": [
"179/da89ab4d",
"179/da89ab4d",
"179/da89ab4d"
]
},
{
"t": 185,
"daa": 187,
"epoch": 3,
"class": 3,
"nodes": [
"187/e0f61041",
"187/e0f61041",
"187/e0f61041"
]
},
{
"t": 200.1,
"daa": 204,
"epoch": 3,
"class": 3,
"nodes": [
"204/5d37e885",
"204/5d37e885",
"204/5d37e885"
]
},
{
"t": 215.1,
"daa": 222,
"epoch": 3,
"class": 3,
"nodes": [
"222/46643165",
"222/46643165",
"222/46643165"
]
},
{
"t": 230.1,
"daa": 236,
"epoch": 3,
"class": 3,
"nodes": [
"236/17eb5a95",
"236/17eb5a95",
"236/17eb5a95"
]
},
{
"t": 245.2,
"daa": 247,
"epoch": 4,
"class": 3,
"nodes": [
"247/7aada19d",
"247/7aada19d",
"247/7aada19d"
]
},
{
"t": 260.2,
"daa": 261,
"epoch": 4,
"class": 3,
"nodes": [
"261/15d8f83e",
"261/15d8f83e",
"261/15d8f83e"
]
},
{
"t": 275.3,
"daa": 274,
"epoch": 4,
"class": 3,
"nodes": [
"274/9f3d19c5",
"274/9f3d19c5",
"274/9f3d19c5"
]
},
{
"t": 290.3,
"daa": 297,
"epoch": 4,
"class": 3,
"nodes": [
"297/fa7975bd",
"297/fa7975bd",
"297/fa7975bd"
]
}
]
}

View file

@ -0,0 +1,367 @@
{
"pass": true,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1f010000",
"template_switch": {
"epoch": 3,
"daa": 181,
"at": 127.9
},
"run_ended_at_s": 280.2,
"final_daa": 300,
"blocks": {
"total": 304,
"before_boundary": 182,
"after_boundary": 122,
"chain_before": 180,
"chain_after": 120
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "8f8806638d59850f",
"seed": "234e082d653dc69d",
"miners": 3,
"disagree": false,
"ready_ms": [
179,
178,
179
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "e145305446a6b4ce",
"seed": "09952ab515cc5a10",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "743ad2a3cab0518a",
"seed": "1812a8eabb5a452c",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "a6523b90cff501e3",
"seed": "6296e3d38df15872",
"miners": 3,
"disagree": false,
"ready_ms": [
191,
191,
191
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "bc811b3c4b8b1ced",
"seed": "a59152f149a517e0",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "e784541f19cdebe5",
"seed": "87ec7849fce0019f",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
}
],
"accepted_per_miner": [
101,
89,
113
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"a9ce45df8beeaf13",
"a9ce45df8beeaf13",
"a9ce45df8beeaf13"
],
"block_counts": [
303,
303,
303
],
"tips_per_node": [
1,
1,
1
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.6,
"daa": 0,
"epoch": 0,
"class": 2,
"nodes": [
"0/234e082d",
"0/234e082d",
"0/234e082d"
]
},
{
"t": 19.7,
"daa": 48,
"epoch": 0,
"class": 2,
"nodes": [
"48/78df0834",
"48/78df0834",
"48/78df0834"
]
},
{
"t": 34.7,
"daa": 67,
"epoch": 1,
"class": 2,
"nodes": [
"67/0b934b49",
"67/0b934b49",
"67/0b934b49"
]
},
{
"t": 49.7,
"daa": 84,
"epoch": 1,
"class": 2,
"nodes": [
"84/2ba5b427",
"84/2ba5b427",
"84/2ba5b427"
]
},
{
"t": 64.8,
"daa": 101,
"epoch": 1,
"class": 2,
"nodes": [
"101/6574b577",
"101/6574b577",
"101/6574b577"
]
},
{
"t": 79.8,
"daa": 123,
"epoch": 2,
"class": 2,
"nodes": [
"123/21e2b6ce",
"123/21e2b6ce",
"123/21e2b6ce"
]
},
{
"t": 94.8,
"daa": 134,
"epoch": 2,
"class": 2,
"nodes": [
"134/74b83d55",
"134/74b83d55",
"134/74b83d55"
]
},
{
"t": 109.9,
"daa": 153,
"epoch": 2,
"class": 2,
"nodes": [
"153/71b251a6",
"153/71b251a6",
"153/71b251a6"
]
},
{
"t": 124.9,
"daa": 175,
"epoch": 2,
"class": 2,
"nodes": [
"175/b1500fb5",
"175/b1500fb5",
"175/b1500fb5"
]
},
{
"t": 139.9,
"daa": 185,
"epoch": 3,
"class": 3,
"nodes": [
"185/dd8ff66c",
"185/dd8ff66c",
"185/dd8ff66c"
]
},
{
"t": 155,
"daa": 191,
"epoch": 3,
"class": 3,
"nodes": [
"191/2ef77979",
"191/2ef77979",
"191/2ef77979"
]
},
{
"t": 170,
"daa": 194,
"epoch": 3,
"class": 3,
"nodes": [
"194/1871cd31",
"194/1871cd31",
"194/1871cd31"
]
},
{
"t": 185,
"daa": 206,
"epoch": 3,
"class": 3,
"nodes": [
"206/808070af",
"206/808070af",
"206/808070af"
]
},
{
"t": 200.1,
"daa": 221,
"epoch": 3,
"class": 3,
"nodes": [
"221/1bccc51c",
"221/1bccc51c",
"221/1bccc51c"
]
},
{
"t": 215.1,
"daa": 237,
"epoch": 3,
"class": 3,
"nodes": [
"237/d0bfaef8",
"237/d0bfaef8",
"237/d0bfaef8"
]
},
{
"t": 230.1,
"daa": 253,
"epoch": 4,
"class": 3,
"nodes": [
"253/a4136a32",
"253/a4136a32",
"253/a4136a32"
]
},
{
"t": 245.2,
"daa": 268,
"epoch": 4,
"class": 3,
"nodes": [
"268/51ef3166",
"268/51ef3166",
"268/51ef3166"
]
},
{
"t": 260.2,
"daa": 277,
"epoch": 4,
"class": 3,
"nodes": [
"277/869d0dd3",
"277/869d0dd3",
"277/869d0dd3"
]
},
{
"t": 275.2,
"daa": 292,
"epoch": 4,
"class": 3,
"nodes": [
"292/fe54acb8",
"292/fe54acb8",
"292/fe54acb8"
]
}
]
}

View file

@ -0,0 +1,474 @@
{
"pass": false,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true,
"metal_prepare_sent_for_v3": true,
"metal_worker_prepared_v3_pack": true,
"metal_no_need_or_mismatch": false,
"metal_accepted_blocks_after_switch": true,
"metal_cpu_recheck_clean": true,
"metal_swapped_without_pause": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1e010000",
"template_switch": {
"epoch": 3,
"daa": 180,
"at": 264.5
},
"run_ended_at_s": 366.8,
"final_daa": 301,
"blocks": {
"total": 305,
"before_boundary": 182,
"after_boundary": 123,
"chain_before": 91,
"chain_after": 67
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "3e974094b5ce543f",
"seed": "130e3b7d568db07a",
"miners": 2,
"disagree": false,
"ready_ms": [
192,
192
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "1d5b129e89ec2995",
"seed": "bc056261dafe0ec2",
"miners": 2,
"disagree": false,
"ready_ms": [
4,
4
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "835e953cdc2c11be",
"seed": "1908705502833bdf",
"miners": 2,
"disagree": false,
"ready_ms": [
9,
7
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "082b7ee882aaf418",
"seed": "4ae011065b84f679",
"miners": 2,
"disagree": false,
"ready_ms": [
1069,
1085
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "7ef81e08e4611665",
"seed": "d5964a745dbe2ce2",
"miners": 2,
"disagree": false,
"ready_ms": [
6,
6
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "ecee6312983c15ee",
"seed": "f9d5de9eea7330ae",
"miners": 2,
"disagree": false,
"ready_ms": [
5,
5
]
}
],
"accepted_per_miner": [
301,
2,
1
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"66a33eca15d33ba2",
"66a33eca15d33ba2",
"66a33eca15d33ba2"
],
"block_counts": [
304,
304,
304
],
"tips_per_node": [
2,
2,
2
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.6,
"daa": 1,
"epoch": 0,
"class": 2,
"nodes": [
"1/97755438",
"0/130e3b7d",
"0/130e3b7d"
]
},
{
"t": 19.7,
"daa": 3,
"epoch": 0,
"class": 2,
"nodes": [
"3/52e2c440",
"3/52e2c440",
"3/52e2c440"
]
},
{
"t": 34.7,
"daa": 3,
"epoch": 0,
"class": 2,
"nodes": [
"3/52e2c440",
"3/52e2c440",
"3/52e2c440"
]
},
{
"t": 49.7,
"daa": 13,
"epoch": 0,
"class": 2,
"nodes": [
"13/fa702087",
"13/fa702087",
"13/fa702087"
]
},
{
"t": 64.8,
"daa": 38,
"epoch": 0,
"class": 2,
"nodes": [
"38/2a687376",
"38/2a687376",
"38/2a687376"
]
},
{
"t": 79.8,
"daa": 55,
"epoch": 0,
"class": 2,
"nodes": [
"55/2f1f0ea8",
"55/2f1f0ea8",
"55/2f1f0ea8"
]
},
{
"t": 94.8,
"daa": 57,
"epoch": 0,
"class": 2,
"nodes": [
"57/e1b2abc1",
"57/e1b2abc1",
"57/e1b2abc1"
]
},
{
"t": 109.9,
"daa": 61,
"epoch": 1,
"class": 2,
"nodes": [
"61/86fa7300",
"61/86fa7300",
"61/86fa7300"
]
},
{
"t": 124.9,
"daa": 61,
"epoch": 1,
"class": 2,
"nodes": [
"61/86fa7300",
"61/86fa7300",
"61/86fa7300"
]
},
{
"t": 140,
"daa": 61,
"epoch": 1,
"class": 2,
"nodes": [
"61/86fa7300",
"61/86fa7300",
"61/86fa7300"
]
},
{
"t": 155,
"daa": 73,
"epoch": 1,
"class": 2,
"nodes": [
"73/e7f6e171",
"73/e7f6e171",
"73/e7f6e171"
]
},
{
"t": 170.1,
"daa": 94,
"epoch": 1,
"class": 2,
"nodes": [
"94/0b1c93b1",
"94/0b1c93b1",
"94/0b1c93b1"
]
},
{
"t": 185.2,
"daa": 114,
"epoch": 1,
"class": 2,
"nodes": [
"114/16034702",
"114/16034702",
"114/16034702"
]
},
{
"t": 200.2,
"daa": 117,
"epoch": 1,
"class": 2,
"nodes": [
"117/d018a345",
"117/d018a345",
"117/d018a345"
]
},
{
"t": 215.4,
"daa": 120,
"epoch": 2,
"class": 2,
"nodes": [
"120/e022d64e",
"120/e022d64e",
"120/e022d64e"
]
},
{
"t": 230.4,
"daa": 134,
"epoch": 2,
"class": 2,
"nodes": [
"134/e7d07203",
"134/e7d07203",
"134/e7d07203"
]
},
{
"t": 245.4,
"daa": 157,
"epoch": 2,
"class": 2,
"nodes": [
"157/d23b3a1b",
"157/d23b3a1b",
"157/d23b3a1b"
]
},
{
"t": 260.5,
"daa": 175,
"epoch": 2,
"class": 2,
"nodes": [
"175/4bf5acbe",
"175/4bf5acbe",
"175/4bf5acbe"
]
},
{
"t": 275.5,
"daa": 198,
"epoch": 3,
"class": 3,
"nodes": [
"198/f337dbc1",
"198/f337dbc1",
"198/f337dbc1"
]
},
{
"t": 290.6,
"daa": 217,
"epoch": 3,
"class": 3,
"nodes": [
"217/dc28325d",
"217/dc28325d",
"217/dc28325d"
]
},
{
"t": 305.6,
"daa": 234,
"epoch": 3,
"class": 3,
"nodes": [
"234/a0079166",
"234/a0079166",
"234/a0079166"
]
},
{
"t": 320.7,
"daa": 249,
"epoch": 4,
"class": 3,
"nodes": [
"249/eb206cb2",
"249/eb206cb2",
"249/eb206cb2"
]
},
{
"t": 335.7,
"daa": 268,
"epoch": 4,
"class": 3,
"nodes": [
"268/9b9866b1",
"268/9b9866b1",
"268/9b9866b1"
]
},
{
"t": 350.7,
"daa": 285,
"epoch": 4,
"class": 3,
"nodes": [
"285/3263ff7a",
"285/3263ff7a",
"285/3263ff7a"
]
},
{
"t": 365.8,
"daa": 299,
"epoch": 4,
"class": 3,
"nodes": [
"299/31eff7d0",
"299/31eff7d0",
"299/31eff7d0"
]
}
],
"metal": {
"worker": "/Users/joshm/Projects/igneum-wt-ca2-v3/proto-metal/igneum-bench-ca2",
"prepares": [
"PREPARE sent for epoch seed b7eda892cff02b39c043f5fa683a7c1629c96d02fa480522603be595372e371f day 1243916 class v2 (6 DAA blocks before the boundary at 60; CPU side ready in 960 ms)",
"PREPARE sent for epoch seed bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c day 1243916 class v2 (the current pair, after seed mismatches; CPU side ready in 1248 ms)",
"PREPARE sent for epoch seed 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3 day 1243916 class v2 (6 DAA blocks before the boundary at 120; CPU side ready in 1768 ms)",
"PREPARE sent for epoch seed 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c day 1243916 class v3 (6 DAA blocks before the boundary at 180; CPU side ready in 1740 ms)",
"PREPARE sent for epoch seed d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 day 1243916 class v3 (6 DAA blocks before the boundary at 240; CPU side ready in 1289 ms)",
"PREPARE sent for epoch seed f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e day 1243916 class v3 (7 DAA blocks before the boundary at 300; CPU side ready in 1166 ms)"
],
"prepared": [
"worker: prepared b7eda892cff02b39c043f5fa683a7c1629c96d02fa480522603be595372e371f 69676e65756d2d6461792f0cfb120000000000 35663.4 program 64.5 dataset 0.0 race 35598.9 variant base class v2 loads/hash 128 cache-fill 2.0 build 30.5 resident 2 programs 1 datasets (prepare answered 1.3 s after it was sent)",
"worker: prepared bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c 69676e65756d2d6461792f0cfb120000000000 33632.6 program 108.1 dataset 0.0 race 33524.5 variant base class v2 loads/hash 128 cache-fill 2.0 build 30.5 resident 2 programs 1 datasets (prepare answered 35.2 s after it was sent)",
"worker: prepared 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3 69676e65756d2d6461792f0cfb120000000000 34728.4 program 77.3 dataset 0.0 race 34651.0 variant base class v2 loads/hash 128 cache-fill 2.0 build 30.5 resident 2 programs 1 datasets (prepare answered 0.0 s after it was sent)",
"worker: prepared 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c 69676e65756d2d6461792f0cfb120000000000 252.5 program 55.5 dataset 196.9 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.3 s after it was sent)",
"worker: prepared d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 69676e65756d2d6461792f0cfb120000000000 51.1 program 51.1 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.1 s after it was sent)",
"worker: prepared f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e 69676e65756d2d6461792f0cfb120000000000 46.0 program 46.0 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.0 s after it was sent)"
],
"prepared_v3": [
"worker: prepared 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c 69676e65756d2d6461792f0cfb120000000000 252.5 program 55.5 dataset 196.9 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.3 s after it was sent)",
"worker: prepared d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 69676e65756d2d6461792f0cfb120000000000 51.1 program 51.1 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.1 s after it was sent)",
"worker: prepared f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e 69676e65756d2d6461792f0cfb120000000000 46.0 program 46.0 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.0 s after it was sent)"
],
"need": 0,
"mismatch_lines": [
"1791239226.967 PACK OUT OF DATE: the prepared pair b7eda892cff02b39 is not the node's epoch bc056261dafe0ec2; rebuilding the program pack for the worker"
],
"refused": [],
"swaps": [
"SEED CHANGE at daa 60: epoch seed 130e3b7d568db07aad638f67b2ac26e7cf93c51e4f564d45c6724fc4b2d3da65 -> bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c, day 1243916 -> 1243916 (epoch 1): the pair was not prepared (unexpected seeds); the worker compiles inline",
"SEED CHANGE at daa 120: epoch seed bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c -> 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3, day 1243916 -> 1243916 (epoch 2): prepare was sent but the worker has not answered prepared yet; it compiles inline",
"SEED CHANGE at daa 180: epoch seed 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3 -> 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c, day 1243916 -> 1243916 (epoch 3): swapped with no pause (prepared 4 s ago, prepare took 252 ms)",
"SEED CHANGE at daa 240: epoch seed 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c -> d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6, day 1243916 -> 1243916 (epoch 4): swapped with no pause (prepared 4 s ago, prepare took 51 ms)",
"SEED CHANGE at daa 300: epoch seed d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 -> f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e, day 1243916 -> 1243916 (epoch 5): swapped with no pause (prepared 6 s ago, prepare took 46 ms)"
],
"accepted_total": 301,
"accepted_after_switch": 124,
"found_lines": 0,
"cpu_recheck_mismatched": 0,
"last_status": ""
}
}

View file

@ -0,0 +1,193 @@
# Counter ASIC 2.0: the node side (program class v3 as a height switch)
5 October 2026, night, worker "ca2-node". Branches: `ca2-v3` (main repository: the igneum-pow seam, the workers, the fast-time gate) and `ca2-v3-node` (the fork, from the 0.3.10 tip 21d4c73c plus pack-loop 05ef0fa3). Plan: `docs/plans/counter-asic-2-rollout.md`; status: `docs/plans/counter-asic-2-status.md`. The shape follows finality v3 (`docs/plans/finality-v3-rollout-devnet.md` section 6): one height switch read from the override file, every node carries the same object before the height.
Everything below is the SEAM. The class itself (`igneum_pow::V3_CLASS`) is a placeholder, w16 (`LoadClass::fixed(4, 16)`), that the integration branch replaces with the decided width, mix and scratch share; the ca2-era draw and the ca2-mixer item construction fill what `Epoch::from_chain_seeds` calls. Nothing here changes a v2 program, a pinned pack or any live node.
## 1. What changed, where
| Piece | What |
|---|---|
| igneum-pow `generator.rs` | `ProgramClass { V2, V3 }`, `V3_CLASS` (placeholder), `GENERATOR_VERSION_V3 = 3`, `generate_from_seed_bytes_program_class(label, seed, class, era)`; `Program::era_bytes`; a v3 program's id is `program_id(3, seed, attempt)` (spec 01 section 1.4.6) |
| igneum-pow `verify.rs` | `Epoch::from_chain_seeds(epoch, day, era, class, label)`, `Epoch::chain_program` (no cache fill), `Epoch::chain_dataset(day, class)` (the one entry the node's day cache goes through; today both classes build the same cache) |
| igneum-pow `emit.rs` | program.h: `IGNEUM_PROGRAM_CLASS "v3"` and `IGNEUM_ERA_SEED_HEX` beside `IGNEUM_GENERATOR 3`; program.json: `program_class`, `era_seed_bytes`; nothing on a v2 pack (`tests/packs.rs` diffs the pinned packs `igneum-genesis-mh` and `igneum-devnet-v4-epoch0`: identical) |
| igneum-pow `packcheck.rs` | `verify_pack_texts_chain` / `verify_pack_dir_chain(dir, epoch, day, want_class, want_era)`: `PackFault::WrongClass` for the wrong class, the wrong era, or a generator other than 2 or 3; `PackIdentity` carries `generator`, `class`, `era_hex` |
| `proto-cuda/nvrtc/packfile.h` | `pf_load` refuses a generator other than 2 or 3 (spec 1.4.5), reads the class (must match the generator) and the era; `pf_pack_class_ok`, `pf_class_token` (the one rule for the `class=` / `era=` tokens) |
| `proto-cuda/nvrtc/worker.cpp`, `proto-opencl/host.c` | a pair's identity includes its class and era when the line names them; right seeds and the wrong class answer `need <e> <d>` and `error <id> pack <dir>: program class mismatch ...`, so the miner prepares the pair from a pack of the right class; a prepare on a wrong-class pack fails in plain words |
| `proto-metal/main.swift` | refuses every `class=v3` line (the Swift generator is version 2; the integration adds 3) |
| fork `consensus/core/src/igneum.rs` | `ProgramClass`, `program_class_for_epoch_at(e, N4, L)`, `program_class_v3_first_epoch_at`, the process-wide activation (`install_program_class_v3_activation`, `program_class_for_epoch`, `program_class_at`), `POW_ERA_BLOCKS`, `POW_ERA_LEAD`, `pow_era_index`, `pow_era_seed_score`; `PowEpochInfo` + `program_class`, `next_program_class`, `program_class_v3_activation_daa`, `era_index`, `era_seed` (serde defaults: v2, never, none) |
| fork `consensus/core/src/config/params.rs` | `program_class_v3_activation_daa` in `Params` and `OverrideParams` (default never on devnet, simnet, mainnet; 0 on the testnet like every other switch), `override_params`, `From<Params>`, the digest (unconditionally, right after `finality_v3_activation_daa`), `Params::install_program_class_v3_activation`, `Params::program_class_v3_first_epoch`; tests `override_params_carry_the_program_class_v3_activation`, the digest test's 11th edit, `fast_time_60x_file_is_the_devnet_at_60x` (every field present) |
| fork `kaspad/src/daemon.rs` | `Program class v3 from the override file: active from epoch E (DAA score N4 rounded up to the epoch boundary at 3600*E, epochs of 3600 DAA)`; the activation installed next to the PoW schedule |
| fork `consensus/pow/src/igneum.rs` | `EpochSeeds { epoch, day, class, era }` (+ `EpochSeeds::v2`), day caches keyed on `(day, class)`, the program through `Epoch::chain_program`, the cache through `Epoch::chain_dataset`, `standalone_epoch` through `Epoch::from_chain_seeds`; test `program_class_v3_seeds_hash_their_own_program_over_their_own_cache` |
| fork `consensus/src/pipeline/header_processor/{processor,pre_ghostdag_validation}.rs` | the class of the header's epoch (`program_class_at(daa)`), `HeaderProcessor::era_seed` (the stand-in, memoised per era), `RuleError::EraSeedUnavailable` |
| fork `consensus/src/processes/pruning_proof/igneum_pow.rs` | `seeds_for`: the class of the epoch; era 0 = genesis; a later era is `PruningImportError::MissingEraSeed` (no era witness in the proof format yet) |
| fork `consensus/src/consensus/mod.rs` | `get_pow_epoch_info`: the class of this and the next epoch, the activation, the era index and seed (the era walk memoised once per era per process) |
| fork `rpc/core/src/model/message.rs`, `rpc/grpc/core/proto/rpc.proto` (fields 12 to 16), `rpc/grpc/core/src/convert/message.rs` | `RpcPowEpochInfo` + `program_class` (generator number), `next_program_class`, `program_class_v3_activation_daa`, `era_index`, `era_seed`; an old node's absent fields read as v2, never, none |
| fork `igneum/miner/src/main.rs` | seeds from the template (`seeds_from_info`), the activation installed from the template, the legacy seed walk keys the class on it; `next_pair` takes the next epoch's class; the job and prepare lines end with `class=v3 era=<hex>` for a v3 epoch; `seeds.txt` carries `program_class` and `era_seed_hex`; `write_pack_checked` checks class and era (`verify_pack_dir_chain`); the "program and 256 MiB cache ready" lines and `export-pack` print the class and the program id |
| `infra/fast-time/override-60x.json`, `tools/finality-attacks/redteam/override-60x-v3.json`, `infra/fast-time/README.md` | the field at never, the README row |
| `infra/fast-time/class-v3.mjs` | the G4 gate (section 5) |
## 2. The epoch-boundary rule
One epoch has one program (spec 01 section 1.12), so the switch keys on the EPOCH: epoch `e` is class v3 when `L * e >= N4` with `L` the live epoch length (`pow_epoch_blocks()`, 3,600 on the devnet, 60 on the fast-time profile). The first v3 epoch is `ceil(N4 / L)`; a height inside an epoch rounds UP to the next boundary and never splits an epoch between two programs. A block's class is a function of its DAA score alone (`program_class_at(daa)`), as its epoch seed is.
| N4 | L | first v3 epoch | first v3 DAA | the epoch before |
|---|---|---|---|---|
| 150 | 60 | 3 | 180 | epoch 2 (DAA 120 to 179) is v2 to its last block |
| 180 | 60 | 3 | 180 | |
| 181 | 60 | 4 | 240 | |
| 136,000 | 3,600 | 38 | 136,800 | epoch 37 (133,200 to 136,799) is v2 |
| 136,800 | 3,600 | 38 | 136,800 | |
| 0 | any | 0 | 0 | (the testnet) |
| never | any | none | | |
The unit tests `program_class_switch_rounds_up_to_the_epoch_boundary` (consensus-core) and `override_params_carry_the_program_class_v3_activation` (params) pin these rows and sweep every activation 0..399 at L = 60.
The digest: the field (and `pow_genesis_dataset_log2`) enters `consensus_digest` unconditionally, so the digest flips the moment a binary carrying the field (at never) runs, exactly as the finality v3 field did. This is intended: every node must carry the object before any node reaches the height, and a node without the field is refused at the handshake (rollout section 1, order step 1). Measured: the devnet digest with no override file moves from `9409dedac4bf9f0f...` (0.3.10) to `c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c` (0.3.11; the pinned value of `consensus_digest_keeps_the_0_3_5_value_until_the_fee_switch_is_set`), which is the expected digest of a scratch node at the publish.
## 3. The era stand-in
`E_n` of spec 04 section 4.4, until the 1-hour VDF is in the node (`docs/plans/era-layout.md` section 2):
| Era | `E_n` |
|---|---|
| 0 | the genesis block hash |
| n >= 1 | the hash of the last selected-chain block whose DAA score is below `15,552,000 n - 7,200` |
One function per path: `HeaderProcessor::era_seed` (the header's era from its DAA score, the walk down the selected parents from the header's selected parent, memoised per era in `era_seed_memo`), `Consensus::get_pow_epoch_info` (the same walk from the sink for the template, memoised once per era per process), `ProofSeeds::seeds_for` (era 0 only). The seed block is at least one era lead (7,200 DAA) below any header that uses it, past the merge depth (3,600), so one walk per era per process is sound; the walk itself is up to an era long on the first header of era `n >= 1` (about 15.5 million selected parents), which is why it is memoised and why the VDF should land before era 1 (180 days after genesis). A v2 program never reads the era; the placeholder v3 class does not either (the ca2-era draw will); the era is carried and recorded in the pack so a worker of the wrong era is refused from the first v3 build.
## 4. The job line, the prepare line, the pack
| Surface | Class v2 (today) | Class v3 |
|---|---|---|
| job line | `job <id> <prehash> <target> <start> <count> <epoch> <day>` | the same with ` class=v3 era=<64 hex>` at the end |
| prepare line | `prepare <epoch> <day> [<dir>]` | the same with ` class=v3 era=<64 hex>` at the end |
| `program.h` | `IGNEUM_GENERATOR 2` | `IGNEUM_GENERATOR 3`, `IGNEUM_PROGRAM_CLASS "v3"`, `IGNEUM_ERA_SEED_HEX "<64 hex>"` |
| `program.json` | `"generator": 2` | `"generator": 3`, `"program_class": "v3"`, `"era_seed_bytes": "<hex>"` |
| `seeds.txt` | `epoch_seed_hex`, `day_seed_hex`, `day_index` | plus `program_class v3`, `era_seed_hex <hex>` |
| template `pow_epoch` | `programClass 2` | `programClass 3`, `nextProgramClass`, `programClassV3ActivationDaa`, `eraIndex`, `eraSeed` |
A v3 program's identity is the pair (program id, era seed): the id covers the generator, the seed words and the attempt (spec 01 section 1.4.6, unchanged), so every era of one epoch seed shares one id, and the era seed, carried by the pack (`IGNEUM_ERA_SEED_HEX`) and the job line (`era=`), tells them apart. The workers and `packcheck` compare both.
A v2 line and a v2 pack are byte for byte what the workers read before this branch (the tokens are sent only for a v3 epoch), so a 0.3.10 worker on a 0.3.11 miner mines v2 epochs unchanged and refuses nothing until the switch; by the switch every worker is 0.3.11 (rollout order).
Refusals: `pf_load` refuses a generator that is not 2 or 3 (`error 0 pack <dir>: program pack generator N is not a generator version this worker runs (2 or 3)`, the exit-44 path of 05ef0fa3: the miner rebuilds the pack before the restart). A job of class v3 against a resident v2 pack of the same seeds answers `need <e> <d>` and `error <id> pack <dir>: program class mismatch: this pack is class v2, the job names class v3 (export the pack again)`; the miner's `need` handling prepares the pair again, `write_pack_checked` writes a v3 pack (checked with `verify_pack_dir_chain` before the worker hears of it), and the prepared v3 pair wins over the resident v2 pair because the pair identity now carries the class. The Metal worker answers `error <id> program class v3 is not implemented by this worker` until the Swift generator carries version 3.
## 5. The fast-time gate (rollout G4)
`node infra/fast-time/class-v3.mjs [--secs 420] [--activation 150] [--epochs-after 2]` under `tools/lock/with-lock.sh run`: three nodes on `override-60x.json` merged with `genesis_bits` 0x1f010000 (2^16 hashes per block, the CPU difficulty of `sim/difficulty/testnet_v2.py`) and `program_class_v3_activation_daa` 150 (inside epoch 2, so the rounding rule is exercised: the first v3 epoch is 3 at DAA 180); one real CPU miner per node (`--engine igneum-pow`, 1 thread, real lottery-hash solutions, every node verifying the other two); the run ends two epochs after the boundary.
### Result, run 1 (5 October 2026, 21:33:04Z to 21:38:02Z, Apple M5 Max shared with other agents' builds)
Binaries: fork ca2-v3-node 79bd8e10 and igneum-pow at ca2-v3 66eeba3 (the mixer-x4 class, `V3_CLASS` = MX4, before the era and hot-table fields), `target-ca2/release`, built on the Mac under the build lock. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2133Z-mx4.json`. PASS: every check true.
| Check | Measured |
|---|---|
| Switch line on every node | 3 of 3: `Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)` |
| Digest, all three nodes | `0186df7d0834d054...` (the 60x profile with the CPU bits and the switch) |
| Template class per epoch | epochs 0 to 2 class 2 (`nextProgramClass` 3 from epoch 2), epochs 3 to 5 class 3; the switch seen at DAA 180, 168.0 s wall |
| Blocks before / after the boundary (node 0's DAG) | 181 / 124 (selected chain 176 / 123), 305 in all |
| Program id per epoch, all three miners agreeing | e0 v2 `8f8806638d59850f`, e1 v2 `fd9562df32a68313`, e2 v2 `1ae6d90ab299154c`, e3 v3 `5d0dedd9fd9e29a1`, e4 v3 `e81808dcdb02ce05`, e5 v3 `06aff9c1d33e7a13`; no v2 id reappears under v3 |
| Rejected blocks | miners 0 / 0 / 0 (97, 103, 104 accepted); nodes 0 / 0 / 0 `PoW rejected` lines |
| Forks | sinks `082fd39ba65df2ff` on all three nodes, block counts 304 / 304 / 304, one tip each |
| Program and cache ready, one CPU core | a new `(day, class)` cache: v2 epoch 0 219 to 235 ms, the first v3 epoch 177 to 185 ms (two per miner: the 24-minute day rolled at DAA 190); a program swap inside a day 2 ms |
The mixer x4 build-time number the rollout asks for is not visible here: the CPU miner derives dataset words on demand from the cache (no dataset build), so the x4 cost lands on the GPU workers' dataset build, measured by the ca2-mixer playbooks on the PCs.
### Result, run 2 (5 October 2026, 21:46:36Z to 21:51:31Z): the composed class
Binaries rebuilt on ca2-v3 b105a55 (era layout merged on the mixer: `V3_CLASS` = MX4 with the era drawn inside `generate_from_seed_bytes_program_class`, hot `None`), fork 79bd8e10 unchanged. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2146Z-era-mx4.json`. PASS: every check true.
| Check | Measured |
|---|---|
| Switch lines, digest | 3 of 3, the same line as run 1; digest `0186df7d0834d054...` |
| Template class per epoch | epochs 0 to 2 class 2, 3 to 5 class 3; the switch at DAA 180, 171.0 s wall |
| Blocks before / after the boundary | 181 / 124 (selected chain 180 / 122), 305 in all; 102, 102, 100 accepted per miner |
| Program id per epoch, all three miners agreeing | e0 v2 `8f8806638d59850f`, e1 v2 `bb8dd9ddbf9eb63f`, e2 v2 `8ee7a9f33d418e48`, e3 v3 `2d278041ba482dba`, e4 v3 `2ae786d294a8a59d`, e5 v3 `bc36813df2f41b5f` |
| Rejected blocks | miners 0 / 0 / 0, nodes 0 / 0 / 0 |
| Forks | sinks `712c1b212091dcdc` on all three nodes, block counts 303 / 303 / 303, one tip each |
| Program and cache ready, one CPU core | v2 epoch 0 178 ms, the first v3 epoch 181 ms, an in-day swap 2 ms |
CPU hash rate across the switch, run 2, miner cpu0 (one thread, the Mac shared with other agents' builds, so approximate): the cumulative rate read 0.024 MH/s through the v2 epochs (30 to 151 s), then fell to 0.021 MH/s cumulative by 271 s (100 s under v3), which puts the v3 interval rate near 0.017 MH/s, about 30 percent under v2 on the CPU interpreter (the era's strided windowed loads and the mixer path). The node has no per-block verify timing line; the CPU verifier is measured in `igneum-pow` (the mixer agent, one M5 Max core, ms per 32-lane unit, same minute, cited from ca2-mixer 1ab8b21's message of 5 October 2026 22:10Z):
| Path | readwidth | ca2-v3 88dafbc (before the fix) | ca2-v3 d233fa1 (after) |
|---|---|---|---|
| v2 (the live devnet) | 0.607 / 0.610 | 1.332 | 0.609 / 0.611 |
| v3 at x4 | | | 1.238 |
| v3 at x8 (the class) | | | 2.077 (worst cold 2.15) |
The regression was `memhard::derive_items` at m = 1 (2.2x); the fix dispatches to an out-of-line `derive_items_mask`, one instance per cache size with the line mask a constant. A v3 block costs the node about 3.4x a v2 block to verify (2.08 against 0.61 ms per unit).
Epoch 0's v2 id is the same in both runs (`8f8806638d59850f`: a v2 program is untouched by the era code, on the chain as in the packs); the v3 ids differ from run 1 because the era draw is now inside the class.
### Result, run 3 (5 October 2026, 22:15:04Z to 22:19:44Z): the final class, x8 with the era
Binaries rebuilt on ca2-v3 d233fa1 (mixer x8, the era drawn inside the class, the verifier fix), fork 89dfcb95. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2215Z-era-mx8.json`. PASS: every check true.
| Check | Measured |
|---|---|
| Switch lines, digest | 3 of 3, the same line; digest `0186df7d0834d054...` |
| Template class per epoch | epochs 0 to 2 class 2, 3 to 5 class 3; the switch seen at DAA 181, 127.9 s wall |
| Blocks before / after the boundary | 182 / 122 (selected chain 180 / 120), 304 in all; 101, 89, 113 accepted per miner |
| Program id per epoch, all three miners agreeing | e0 v2 `8f8806638d59850f`, e1 v2 `e145305446a6b4ce`, e2 v2 `743ad2a3cab0518a`, e3 v3 `a6523b90cff501e3`, e4 v3 `bc811b3c4b8b1ced`, e5 v3 `e784541f19cdebe5` |
| Rejected blocks | miners 0 / 0 / 0, nodes 0 / 0 / 0 |
| Forks | sinks `a9ce45df8beeaf13` on all three nodes, block counts 303 / 303 / 303, one tip each |
| Program and cache ready, one CPU core | v2 epoch 0 179 ms, the first v3 epoch 191 ms (x8 construction: the cache fill is unchanged, the items are derived on demand), an in-day swap 2 ms |
Epoch 0's v2 id `8f8806638d59850f` is the same in all three runs. Three runs, three v3 classes (mixer x4; x4 with the era; x8 with the era), the same chain behaviour each time: the switch rounds up to epoch 3, no block rejected, one chain.
### Result, run 4 (5 October 2026, 22:25:22Z to 22:31:28Z): a real Metal miner across the boundary (gate G4b)
The fleet-outage case: the three runs above used CPU miners, so the Metal worker's class v3 path had not mined. Run 4 puts node 0's miner on the Metal worker the way the app does (`igneum-miner --worker igneum-bench --prepare-packs <dir> --exit-on-seed-change`, `igneum-bench` built from ca2-v3 00c55aa: a class v3 program comes from the pack's `program_bound.metal`, its day from the pack's `memhard.metal` (the x8 construction, the era layout), keyed by (day, class, era)), genesis bits 0x1e010000 (2^24 hashes per block), CPU miners on nodes 1 and 2. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2225Z-metal-mx8.json`.
| Check | Measured |
|---|---|
| The chain | 182 / 123 blocks across DAA 180, 0 rejected (miners and nodes), sinks `66a33eca15d33ba2` on all three, 304 / 304 / 304, 3 of 3 switch lines, the switch at DAA 180, 264.5 s wall |
| PREPARE lines | 6 (three class v2, three class v3), each 6 to 7 DAA before its boundary, the v3 ones with the pack directory and `class=v3 era=<hex>` |
| The worker's `prepared` for the v3 packs | three: `prepared <seed> <day> 252.5 program 55.5 dataset 196.9 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1` (the first, with the day built from the pack: 0.9 ms cache fill, 41.1 ms build on the GPU), then 51.1 and 46.0 ms (the day resident) |
| The swap at the boundary | `SEED CHANGE at daa 180 ... (epoch 3): swapped with no pause (prepared 4 s ago, prepare took 252 ms)`; the same at 240 and 300 |
| Blocks on v3 | 124 accepted after the switch (301 in the run), every one re-checked on the CPU: `mismatched 0`; `need` 0; no class or era mismatch line; no refusal, no exit 42 or 44 |
One line before the switch, on the v2 path: `PACK OUT OF DATE: the prepared pair b7eda892... is not the node's epoch bc056261...; rebuilding the program pack` at DAA 60: the epoch-1 seed reported at `boundary - lead` flipped (the quarter-lead confirm is 3 DAA on the 60x profile, 150 on the devnet) and the miner's rebuild path of 05ef0fa3 wrote the right pack. The v2 prepares answered 33 to 35 s after they were sent (the Metal variant race runs inside the prepare, longer than the 10-DAA lead of the profile; 600 s on the devnet), so epochs 1 and 2 compiled inline; a pack program races nothing and answered in 252 ms. The v3 path is clean; the script now judges the Metal checks from the first v3 prepare on (the run's own report flagged the v2 line and read FAIL on that one check).
The PC 2 suite jobs for the same fork (79bd8e10), the G6 evidence (the coordinator's status file carries the SUMMARY lines):
| Job | Tree | What | Result |
|---|---|---|---|
| `build-20261005-215219` | main b105a55 | the Linux node, the six node suites, the app tests | Linux build and app tests passed; `kaspa-consensus` failed on the known flake (`ban_is_decided_by_the_carrying_block`, `UnexpectedDifficulty` in `mine_on_all`, the 0.3.10 cut's section 11 case) |
| `build-20261005-215712` | main b105a55 | `kaspa-consensus` alone | 97 passed, 0 failed, 3 ignored, `ban_is_decided ...` ok, 21:59:45Z |
| `build-20261005-220351` | main 8ea6740 | the other five crates (`kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows`) and the app tests | 22:04 to 22:06:55Z: igneum-exec 17, igneum-miner 18, consensus-core and p2p-flows green, the app 78 + 26 + 8; `kaspa-pow` FAILED on `program_class_v3_seeds_hash_their_own_program_over_their_own_cache` (the fork test compared a chain v3 program's class to `V3_CLASS` with `era: None`; on the era-merged crate the class carries the drawn era inside). Fixed on the fork at 89dfcb95 (the class minus its era is `V3_CLASS`, the era is drawn, another era seed keeps the program id and hashes another program over the same day cache). Re-run: job 4 |
| `build-20261005-221237` | main d233fa1, fork 89dfcb95 | the five crates and the app | 22:12:37 to 22:15:44Z (185 s): every stage ok; kaspa-consensus-core 108 (2 ignored), igneum-exec 17, kaspa-pow 33 + 14 (the v3 engine test, with the igneum-pow feature), igneum-miner 18, kaspa-p2p-flows 7, igneum-app 78 + 26 + 8; 0 failed. With job 2, G6 is green |
The PC's test stage builds the fork's test binaries with the plan's feature set, so the v3 engine test DID run on PC 2 (the earlier reading that it was Mac-only is withdrawn): the PC job is the G6 evidence.
## 6. Tests
| Where | What | State |
|---|---|---|
| igneum-pow `cargo test --release` | 42 unit + 11 pack tests, including `program_classes`, `program_class_and_era_are_checked`, the pinned-pack diffs | pass (Mac, 5 Oct 2026) |
| `proto-cuda/nvrtc/emu/packfile-test.sh` | 13 checks: the attempt rule, the generator rule, a v3 pack with its era, the token matcher | pass (Mac) |
| `host.c`, `worker.cpp` (emulation), `main.swift` | syntax / compile | pass (Mac) |
| fork `cargo test -p kaspa-pow --features igneum-pow` | 14 engine tests including `program_class_v3_seeds_hash_their_own_program_over_their_own_cache` (rewritten at 89dfcb95 for the era-in-class rule) | pass (Mac, 89dfcb95 against d233fa1; PC 2 job 4: 33 + 14) |
| fork `cargo test -p kaspa-consensus-core` | 108 + 7: the switch rounding, the era clock, the params, digest and fast-time file tests | pass (Mac, fork 79bd8e10) |
| PC 2 `build-job.mjs` suites | `kaspa-consensus-core igneum-exec kaspa-pow kaspa-consensus igneum-miner` + `igneum-app` | the coordinator publishes on its go |
## 6a. The integration merges for the ship (5 October 2026, 22:35Z to 22:45Z)
| Merge | Commit | Conflicts, and how they were kept |
|---|---|---|
| readwidth 30ff674 (the OpenCL `__local` declaration rule, the read-width decision record, the per-watt rows) | 3566afd | `emit.rs` (2 hunks: the hot-table kernel arguments, HEAD's superset), `packbench.swift` (readwidth's resident-footprint lines added to HEAD's hot-aware RESULT line), `bench-log.md` (both entries), `read-width.md` add/add (readwidth's version, a pure superset of HEAD's) |
| origin/master 1f0d62c (the 0.3.10 merge) | 49c7e78 | `host.c` (one struct hunk: both fields, `dupOf` and `pci[32]`; master's topology, duplicate-platform and read-back code and ca2-v3's class/era tokens, `mh_word`, hot and mixer fields all auto-merged; `cc -fsyntax-only` clean), `bench-log.md` (both entries) |
Checks after the merges (49c7e78, 22:50Z): the igneum-pow suite 53 + 4 + 19 + 7 pass; the packfile test 0 failures; the CUDA emulation `emu/test.sh` PASS (ready + prepare 1, 64 + 64 + 32 found on pack A, pack B prepared with its self-test, the swap, the self-heal rebuild of pack A, 17 sampled hashes equal to `igneum-pow hash-bound`, the NVRTC source check PASS for 6 files); `proto-opencl/test-generic.sh` on Apple OpenCL PASS (the same protocol, job 5 refused after the swap, 15 sampled hashes equal); the fork's `kaspa-pow`, `igneum-miner`, `kaspad` check clean. The emulation needs `IGNEUM_CUDA_INC` pointed at a checkout's `proto-cuda/nvrtc/redist/include` (the worktree has none).
## 7. Unverified, and what is owed
- The era walk for era >= 1 has never run (the devnet is 180 days from era 1); a pruning-proof sync past era 0 fails with `MissingEraSeed` until an era witness exists in the proof format.
- The Metal worker takes class v3 from the pack (program and day) and mined across the boundary in run 4; the app must pass `--prepare-packs` to its Metal worker (the app branch fixes `engine.rs`, which passed it to the non-Metal workers only).
- The GPU workers' class refusal was checked by the C test of `pf_pack_class_ok` and the syntax of both hosts, not by a live worker on a v3 pack: the integration's bit-exact gate (G1) is where a real worker first builds a v3 pack.
- `next_pair` keeps the current era seed for the next epoch; an epoch boundary that is also an era boundary (once per 180 days) would prepare the wrong era, and the job line then names the right one, so the worker refuses the prepared pair and the miner prepares again (one wasted compile, no wrong block). The VDF era seed replaces the stand-in before this matters.
- Done 22:05Z: ca2-cache rebased as 1950661 fast-forwarded (the first attempt, 2de19e5 on 464d6e1, conflicted in 8 files and was aborted); igneum-pow 53 + 19 tests and the packfile test pass; the fork's kaspa-pow, miner and kaspad check clean against the merged crate (seam unchanged). `V3_CLASS = { era: None, hot: None, ..LoadClass::MX4 }`.
- Done 22:12Z: ca2-mixer 16dfd1e (MX8 candidate, tests) merged at 4e733bb with two one-line fixes the merge needed (3a7fba7 `MX8` gets `era: None, hot: None`; 795472e tests/scratch.rs's `Instr` literals get `win: 0, off: 0`); then ca2-mixer 1ab8b21 (the verifier fix, `V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }`: class v3 is x8 by the coordinator's decision of 22:05Z) merged at d233fa1. Checks on d233fa1: igneum-pow 53 + 4 + 19 + 7, packfile 0 failures, the fork's `cargo test -p kaspa-pow --features igneum-pow` 14 pass, the release rebuild 1 min 57 s.
- Owed (0.3.12, coordinator's ask of 5 October 2026 22:20Z): per-day dataset reuse in the CUDA and OpenCL workers (a `Day` object shared by consecutive pairs, the cache freed after the build); the Metal worker already keys datasets by day. Until then the iGPU tier mines v3 with a dataset rebuild per epoch on those two workers.
- Wire compatibility: `RpcPowEpochInfo` gained five fields in its Borsh form (wRPC) and five proto fields (gRPC); the gRPC side reads an old node's zeros as v2 / never / none; the Borsh form is versioned by `GetBlockTemplateResponse` (version 2 carries the whole struct), so a 0.3.11 wRPC client against a 0.3.10 node reads short: the miner uses gRPC, the console reads JSON (serde defaults).

View file

@ -0,0 +1,49 @@
# Counter ASIC: the public description in four levels
Josh, 5 October 2026 (night): "not an information overload". Four levels; the layer names appear only from level 3 down, next to their numbers. Numbers come from the final table of `docs/plans/counter-asic-2-status.md`; a number still owed is marked `[owed: ...]`, never guessed. Josh's copy law throughout.
## Level 1: one sentence (site hero, litepaper abstract)
Built for graphics cards. A custom chip gains under 2x, and the model is public. (The bounty is named only once it is escrowed: docs/plans/funding.md rule 3; D11 for Josh.)
## Level 2: one site card, one short litepaper section
Three ideas, no layer names, no widths, no SRAM.
**The hash rewrites itself.** A new program every hour, drawn from the chain. Its memory pattern changes with it. The rules change on a schedule fixed at launch. No release, no vote.
**It waits on memory, not maths.** Every hash is a chain of random reads into a table too big for a chip to carry. The wait is the same physics for everyone.
**Miners hold the switch.** Spare defences are written into the rules, switched off. A 90% miner signal turns one on. No fork.
A custom chip gains under 2x. Model published; a bounty follows the external review. [link: the numbers page]
Litepaper only, a fourth paragraph: No hash has stayed free of chips forever. Igneum does not claim to. It claims the gain is small, the response takes a week, and both are measured.
Site card placement: the Mine section of `site/index.html` beside "no chip can be built for it" (which this card replaces: the claim is a bounded gain, not impossibility). Litepaper placement: `site/litepaper.html` section `mining`, replacing the paragraph that begins "Everything above is automatic" and the "What Igneum does not claim" line on chips; the vs RandomX table keeps its rows, with the "Changes over time" row's Igneum cell reading "A new program every hour, its memory pattern and read widths with it; era draws and reserved families on a schedule fixed at genesis".
## Level 3: the numbers page (`site/bench.html`, section "Counter ASIC")
Headline of the chip model (5 October 2026, night): the strongest chip holds the whole 256 MiB cache on-die (about 128 mm^2 and $46 of silicon at N5 by shipped cache-die density, approximate) and computes dataset items on the fly; its gain over the RTX 5090 is 2.4x as the parameters stand, and no write-scratch share within an 8 GB card's budget changes that. The lever that does is the dataset item's mixer cost (x4: 1.8x with a 3x fixed-function factor, verifier 1.6 to 4.8 ms per warp). Decided 5 October 2026 (delegated): the mixer x4 and the cache growth rule enter class v3, so the headline row is the on-die-cache chip against v3 with everything combined. [owed: the combined row from docs/analysis/chip-model-v3.md; if it reads 1.8x, the claim is "under 2x" with the margin stated as thin, and the next levers are named: the mixer x8 and the hot table.]
Per card, the bench table: the v2 class and the v3 class, hash rate, bytes per hash, the latency-bound share (rate over the card's random-read ceiling per load), the CPU verifier per warp, with machine, date and command. The chip model before and after Counter ASIC 2.0 (the m16 model's gain arithmetic at the v2 class and at the v3 class, with the SRAM a mirror needs, cited or approximate as the analysis says). The bounty terms (spec O-1.17: the leaderboard by card model, the standing bounty for any chip design beating a GPU by more than 2x, January 2027). Here the layers are named next to their numbers: read width, per-program mix, scratch, era layout, working set, hot table, cache schedule, the reserved integer-matrix family.
| Card | v2 MH/s | v3 MH/s (era packs, six eras) | Bytes per hash | Latency-bound share | Verifier ms per warp (v2 / v3, one loaded M5 Max core) | Daily 1 GiB build (v2 / v3) |
|---|---|---|---|---|---|---|
| Apple M5 Max (Metal) | 27.68 | 27.85 to 27.98 (spread 0.5%, the final class) | 512 | 1.06 | 0.61 / 2.08 (3.4x; worst cold 2.15) | 21 / 21 ms |
| RTX 5090 (CUDA) | 137.2 | 135.90 to 137.70 (spread 1.3%, the final class) | 512 | 1.01 | the same verifier | 25 / 23 ms |
| RX 9070 XT (OpenCL) | 18.09 | 18.59 to 19.18 (spread 3.1%, the final class) | 512 | 0.95 | the same verifier | 74 / 75 ms |
Every number measured 5 October 2026 (`docs/plans/era-layout.md`, `docs/plans/mixer-x4.md`, `docs/bench-log.md`); the v3 verifier figure is the x8 mixer on one core at load average 5.5 (the fixed crate; the same session matched readwidth's quiet v2 figure within 1%). Bit-exact: every v3 pack's fingerprint equal on the three vendors.
Chip model, before and after (`docs/analysis/chip-model-v3.md`): the on-die-cache recompute chip (the whole 256 MiB cache in SRAM, about 128 mm^2 and $46 of silicon at N5 by shipped cache-die density, approximate) against the RTX 5090's measured 136.1 MH/s at 50 T integer op/s: class v2 333 MH/s, 2.4x; class v3 (mixer x8) 41.7 MH/s, 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon. The claim "under 2x" holds with the margin stated: 8% on the allowance (a 3.3x allowance reads 1.0x), 9% on the budget. Next levers, named: the mixer at x16 (the verifier at about 4 ms per warp, inside the 10 ms gate; a 2019-class core unmeasured), a hot table small enough to stay resident beside the streaming dataset (measured and not adopted tonight: 32 to 96 MiB tables cost the GPU 7 to 20% and help the chip).
Levers measured and not adopted (5 October 2026): wider reads (16 and 64 B: no card gains, the 5090 goes bandwidth-bound at 64 B), the per-load width mix (5.5 to 22.3% spread), a per-warp write scratch (the chip keeps it implicitly: 2.4x at every share), the hot table (above). Reserved, switched off: the integer matrix family R1 (mm8, native on all three vendors as a tile; dp4a 1.17x a step on the 5090, 1.06x on the 9070 XT, emulation 1.6x on Apple) and the epoch length (600 s to 2 hours by 90% signal; a per-program FPGA bitstream mines 0% of a 600-s epoch).
AMD RDNA 4 sits at about a seventh of a 5090 on this hash by its dependent-read rate (2.4 G against 17.5 G reads per second), 2.2x worse per pound at list prices and 4.9x worse per watt (read-width.md section 4.1, approximate); the card's memory system, not a tuning gap.
Chip model, before and after: [owed: from docs/analysis/sram-mirror.md after the shipped-density correction (256 MiB on-die at about 130 to 165 mm^2 by AMD 3D V-Cache and TSMC N5 macro density, 54 to 83 mm^2 bit-cell-only lower bound), the hot-table and scratch analyses; the on-die-cache recompute chip is a named row per variant].
## Level 4: the analysis documents
`docs/plans/counter-asic-2.md` (the plan and the layer table), `docs/plans/read-width.md`, `docs/plans/era-layout.md`, `docs/plans/hot-table.md`, `docs/analysis/scratch-soundness.md`, `docs/analysis/sram-mirror.md`, `docs/analysis/int8-matrix-family.md`, `docs/analysis/m16-recompute-attacker-2026-10-05.md`, the bench log entries of 5 October 2026 (night), the specification sections 1.4 to 1.13.

View file

@ -0,0 +1,131 @@
# Generator class v3 on the live devnet: rollout plan (5 October 2026, night)
Scope: the DEVNET only. Josh delegated the three decisions for the devnet before going to bed (5 October 2026, about 20:10 UTC, through the coordinator): "Counter ASIC 2.0 fully deployed" tonight. The public testnet is not open; its genesis takes v3 from day one. The devnet is ours and a reset is acceptable.
Shape and rules follow `docs/plans/finality-v3-rollout-devnet.md` and the publish record `docs/plans/finality-v3-devnet-publish.md`: one height switch read from the override file, every node carries the same object before the height, the PCs get igneumd only through an OTA app version, the activation height leaves at least three hours from the manifest publish. Nothing in this file has run on the devnet. The numbers marked `<...>` are filled by the integration branch `ca2-v3` and the decisions of section 6; the plan is published with them, not before.
## 1. What changes and what does not
Only the lottery hash's program class changes, and only from the first epoch at or above the height. One switch, `program_class_v3_activation_daa`, in `Params` and `OverrideParams` like `difficulty_v2_activation_daa`; default `u64::MAX` (never) on every network. Because one epoch has one program (spec 01 section 1.12), the switch keys on the EPOCH: epoch `e` is class v3 when `3,600 e >= N4`, so the activation is rounded up to an epoch boundary and a block's class is a function of its DAA score alone, as today.
Class v3 = generator version 3: the width rule `<W>` (layer 1 or 2, decision 1), no scratch (layer 3 decided out: scratch share 0), the era draw of the table layout and the working set (layers 4 and 8), the hot table of `<S>` MB from the epoch seed (layer 5), the cache growth rule of layer 6 (option C) and the M16 mixer x4 in the dataset item construction (decided 5 October 2026, delegated). New program id (`generator = 3` in the id's preimage, spec 1.4.6), new packs and vectors, new `IGNEUM_GENERATOR` in every pack, a pack of the other version refused by every implementation (spec 1.4.5 already says so).
What does not change: the chain, the genesis, the databases, the day key and the 256 MiB cache fill, the dataset items (spec 1.8.5, if the era interleave keeps the item values; the era-layout document says what it costs otherwise), finality, fees, proving. The SP1 guest does not read the lottery hash (`proving/igneum-prove` has no dependency on `igneum-pow`; the pinned guest of DAA 210,000 is a fee-table switch), so no new guest is pinned. The node's `EpochSeeds` gains the class and the era bytes; `IgneumEngine::epoch_for` builds the v3 `Epoch` from them; the miner's `export-pack` and the serve protocol's job line carry the class so a GPU worker regenerates the right pack from the seed bytes.
The consensus digest (`Params::consensus_digest`, ledger X18) covers every activation height, so the new field enters the digest and every node must carry the same object before any node reaches the height; a node without the field is refused at the handshake once the others carry it, which is the protection the digest exists for. Binary rollout first (the digest flips when the binary carries the field at `never`), the height second.
## 2. The binaries
Built from `<node branch>` at `<commit>` on the PCs through `tools/build-job.mjs` (standing rule 5 October 2026), the Mac binary under the build lock. The table is filled at build time: platform, path, sha256, how it was verified (`--version`, the switch's first line on a private suffix, `strings` carries the field name).
## 3. The activation height N4, and how every node learns it
`N4 = DAA at the manifest publish + 10,800` at least, chosen as `DAA now + 14,400` rounded up to the next epoch boundary (a multiple of 3,600), checked at publish (`N4 - DAA >= 10,800`). The packaged line carries every switch:
```
NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":N4,"proving_v1_activation_daa":N5,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000}'
```
The same nine-field object goes verbatim into the override files of Mac node 1, the observer node and the seed, and into the manifest's `consensus.override` (`publish-manifest.sh --override`). N5 = DAA at publish + 14,400 (the proving v1 switch, no rounding). Two digests, read on the 0.3.11 Mac node (igneumd bd7f043c..., fork 89dfcb95, 22:5x UTC): with no override file c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c (the rolling-upgrade digest, equal to the node agent's pinned test); with the nine-field object at N4 = N5 = 154,800 the ACTIVATION digest 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888, the value every node must print after the publish; the node logs "Program class v3 ... active from epoch 43 (DAA 154800 ... epochs of 3600)" and "Proving v1 ... paid from DAA 154800, 8 blocks a segment, unproven after 600 DAA, aggregator share 1000 bps". Binaries from release-0.3.11 23bc2b2: Igneum-Miner-0.3.11.dmg b7e81d4f6f3af9f9179e29faa72c844b795df757dfb2e4cd1d56cd454e78e1e7 (41,592,041 bytes; the packaged json carries the nine fields, read back from the image); the seed's Linux igneumd 63cf490d... (glibc 2.34, 89dfcb95 inside). The era seed for the devnet: the stand-in of `docs/plans/era-layout.md` (the hash of the last selected-chain block below `15,552,000 n - 7,200`; era 0 on the devnet uses the genesis hash), until the 1-hour VDF of spec 4.4 is in the node.
## 4. The order
1. The digest flip: a node build that carries `program_class_v3_activation_daa` at never on every node (hand nodes and the seed first: `infra/devnet/restart-hand-nodes.sh '<object without the new field>'`, then the app version through the manifest; every node prints the same `Consensus params digest`).
2. Fix N4, cut the app version (`packaging/mac/packaged-config.sh`, the three version files), commit as igneum-josh.
3. The Windows payload inputs (`packaging/windows/push-inputs.sh`) and the Mac DMG (`packaging/mac/build-dmg.sh`), the manifest (`publish-manifest.sh --activation-height N4 --deadline-note "program class v3" --override '<object>' --deploy`), `fetch-ci-artifacts.sh --deploy`.
4. The observer, the seed, Mac node 1 with the object; each node's first lines show every switch and `Program class v3 from the override file: active from epoch <N4 / 3,600>`.
5. HiveOS: `packaging/hive/make-hive-package.sh` republished with the v3 `igneum-miner` and workers, same version string as the apps.
6. The watch: before `N4 - 1,800` both PCs on the new app version (STATUS lines); at the boundary every miner's `prepare` of the v3 pack (the hot-swap entry's shape) and the first v3 block's program id on the observer; the hash rate per card against the measured v3 numbers of `docs/plans/counter-asic-2.md`'s table; zero `pack refused` lines; the CPU verifier time per block in the node log against the measured ms per warp.
### 4a. Two rules for the publish and every PC job (C32)
- The 0.3.11 update-now goes to PC 2 only after the prover-floor agent's server build (floor-build-3, under /opt/igneum-floor in WSL2, published 22:27 UTC, 25 to 90 minutes) has closed: an app restart ends the running job. The update-now takes a machine list (as 0.3.10's did): the Mac, the laptop and PC 1 first, PC 2 last.
- Re-fetch after every app update: an app update clears the jobs folder (the 0.3.10 install at 21:49 UTC took PC 1's AMD kit with it), so every fetch-then-run pair re-publishes its fetch after an update, and every run playbook opens with a presence check of its kit that fails with "kit missing: republish the fetch after the app update".
- No PC job raises an elevation prompt on either PC for the rest of the night (C35): both unexplained app quits tonight came 20 to 41 s after an administrator prompt beside the running installed app (PC 2 20:00:49 to 20:01:09Z, PC 1 22:30:25 to 22:31:06Z), the engine has no self-relaunch after a quit, and nobody is at a keyboard; the two-minute test of that class (raise one prompt from a job while the app mines, cancel it, read the quit line, which ember-tune b671c8b stamps with its source) runs in the morning with Josh present, or tonight only after the relay relaunch path is proven and the rollout is done. The sweeps and Ember's run stay held; the floor, aggregation-cost and M16 jobs raise no prompt.
## 5. Rollback
Before N4: remove the field on every node and restart; nothing has happened (the digest flips back, so every node at once). After N4: there is no rollback by restart, because blocks mined under v3 verify only under v3. A rollback is a second height switch back to v2 at a later epoch, carried the same way. This is why the measurements of the plan come first.
## 6. The decisions, by Josh's rules (devnet)
Josh's rules, applied by the coordinator and recorded here with the number that decided each:
| Decision | Josh's rule | Choice | The number |
|---|---|---|---|
| Width (layer 1) | the widest read that keeps every card we own latency-bound (achieved loads within 90% of the probe ceiling) with margin on the 5090 (its bytes per hash under a third of its bandwidth at the measured rate) | DECIDED (5 October 2026, delegated): keep v2, 128 x 4 B. w16 passes the rule (shares 0.90 / 0.84 / 1.03, 18% of the 5090's stream) but closes nothing and does not move the chip row; w64 and w64x4 make the 5090 bandwidth-bound (share 0.58 / 0.56, 37% of stream) | `docs/plans/read-width.md` (readwidth e752fc7): v2 5090 136.1 MH/s, 9070 XT 18.15, M5 Max 27.74 (gap 7.5x); w16 139.8 / 17.90 / 28.26 (gap 7.8x); w64 71.9 / 17.59 / 28.27 (gap 4.1x); the 9070 XT does 2.4 G dependent reads/s at every width |
| Per-load mix (layer 2) | in, if the min-to-max spread across six programs is under 5% per card | DECIDED (5 October 2026, delegated): out | spreads of the median over six programs: mix 50/35/15 5090 18.8%, 9070 XT 7.4%, M5 Max 11.3%; mix 25/50/25 22.3% / 5.5% / 8.1% |
| Scratch share (layer 3) | the smallest share at which the chip model's gain falls under 1.5x at the lowest GPU cost, within the 6 GB working-set cap | DECIDED (5 October 2026, delegated): 0. No share under the cap moves the on-die-cache recompute chip, so layer 3 is not adopted into v3 | `docs/analysis/scratch-soundness.md` (ca2-soundness a465881): chip 333 MH/s against the 5090's measured 139.7 = 2.4x at 0% RMW; 2.4x at 12.5 / 25 / 50% replaced (chip 381 / 443 / 661 against 160 / 186 / 279 projected) and 2.4x or more added, at 32 and 128 KB; the chip keeps the scratch implicitly in 80 to 320 B per lane because the verifier resets it per unit |
| Activation height N4 | devnet tip + 14,400 at publish, checked >= 10,800 at publish, rounded up to the epoch boundary | `<at publish>` | |
### 6a. The chip model's headline, and how the scratch share is chosen
The layer 6 finding changes the headline: the strongest chip holds the whole cache on-die (about 130 mm^2 at N5/N3E and 165 mm^2 at 7 nm by shipped-product SRAM density, AMD 3D V-Cache 1.56 MB/mm^2 and TSMC N5 HD macros; 54 to 83 mm^2 is the bit-cell-only lower bound; `docs/analysis/sram-mirror.md` after the 20:18 correction) and computes dataset items on the fly through M16's mixer. The before-and-after table must carry that chip as a named row against the GPU for each variant, and the scratch share is chosen by that row: the smallest share at which the on-die-cache chip's gain falls under 1.5x. Result (20:23 UTC): no scratch share under the 6 GB cap gets that chip under 2x; the gain is 2.4x at every share. The site's "under 2x" claim is therefore qualified until the mixer multiplier or the cache rule closes it: M16's mixer multiplier x2 gives 1.2x (3.6x with a 3x fixed-function factor) at 0.8 to 2.4 ms verify per warp, x4 gives 0.6x (1.8x with the factor) at 1.6 to 4.8 ms, inside the 10 ms gate, with the 5090's daily dataset build at 27 and 54 ms. DECIDED (5 October 2026, delegated under "execute the full 2.0 plan" and "deploy what is absolute best"; Josh confirms for the public testnet genesis): M16 mixer x4 goes into v3 behind the same activation. Numbers: attacker 0.083 Ghash/s at 50 T op/s (0.36x bare against 229 MH/s; 1.8x against the 5090's measured 139.7 MH/s with a 3x fixed-function factor, approximate); verifier 1.6 to 4.8 ms per warp (inside the 10 ms gate); the 5090's daily dataset build 54 ms (13.4 x 4, measurement owed). Layer 3 stays out (scratch share 0); its soundness document and pack-contract tests are kept because the construct is sound and may return. The chip model's headline row becomes the on-die-cache recompute chip against v3 with everything combined (x4 mixer, the hot table, the width rule, the era draws, the cache growth), fixed-function factor included; the public level 3 shows that row: if it reads 1.8x the claim is "under 2x" with the margin stated as thin and the mixer x8 and the hot table named as the next levers. The dataset and cache vectors are re-cut once for v3 (cache growth rule and x4 together), the soundness suite re-run on the new construction, bit-exact on the three cards, the verifier per-warp time measured on the Mac; the daily dataset build time quoted for the 5090, the Mac and the 9070 XT. Branch ca2-mixer carries it. Refinement (coordinator, 21:28 UTC, delegated under "as strong as the measurements allow"): x8 is built and measured beside x4 on the same packs (verifier per warp on one Mac core, the 1 GiB daily build on the 5090, the M5 Max and the 9070 XT or gfx1036, the chip row at equal silicon with the 3x factor); x8 goes into v3 if the per-warp verify stays under 10 ms on one core and the daily build stays under 1 s on every card we own, otherwise x4 with the thin margin stated in level 3 and x8 named as the next lever. DECIDED (5 October 2026, 22:06 UTC, delegated): x8. Both halves pass: the per-warp verify at x8 is 2.79 ms on a loaded M5 Max core (2.1x v2; about 1.3 ms quiet, approximate) against the 10 ms gate; the daily 1 GiB build does not move with the mixer on any discrete card (RTX 5090 23 to 25 ms, RX 9070 XT 72 to 77 ms, M5 Max 21 ms at x1, x4 and x8: latency-bound), 13x to 40x under the 1 s bar (PC 1 jobs fetch-mixer-x4-20261005 and run-mixer-x4-pc1-20261005, 22:00:08 to 22:04:39Z, both cards restored, every pack's fingerprint equal to the Mac's). V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }. Chip row at x8: 1,198,080 ops per hash, 41.7 MH/s at 50 T op/s, 0.31x bare, 0.92x with the 3x fixed-function factor, 0.76x at equal silicon: the claim reads under 1x with the factor, margin 8% on the factor and 9% on the budget (chip-model-v3.md). Verifier on the fixed crate (ca2-mixer 1ab8b21, 22:07 UTC, same input beside readwidth's binary, load 5.5): v2 0.609 / 0.611 ms per unit (readwidth 0.607 / 0.610), x4 1.238 / 1.237 (2.0x, worst cold 1.40), x8 2.077 / 2.058 (3.4x, worst cold 2.15): 4.8x inside the 10 ms gate on this loaded core. The vectors are re-cut once on this class. Cost: pool shares per core and IBD time scale with the verifier (x8: 2.1x v2); the integrated tier mines v3 with a restart per epoch until per-day dataset reuse lands (0.3.12). The scratch-soundness analysis carries this table (`docs/analysis/scratch-soundness.md`, question 2).
### 6b. The user tiers (the consequences rule)
AMD RDNA 4 (the RX 9070 XT) sits at about a seventh of an RTX 5090 on this hash by its dependent-read rate (2.4 G against 17.5 G reads per second at 1 GiB, measured tonight), 2.2x worse per pound at list prices (0.032 against 0.072 MH/s per pound, approximate) and 4.9x worse per watt (read-width.md section 4.1). This is the card's memory system, not a tuning gap: no read width closes it without making the 5090 bandwidth-bound. The level 3 numbers page states it.
## 7. Gates before any publish (all of them, no exceptions)
| # | Gate | Evidence required | State |
|---|---|---|---|
| G1 | bit-exact v3 on all three vendors against the Mac reference | GREEN on the final class (job run-ca2-era-pc1b-20261005, 22:16 to 22:21Z, exit 0 in 304 s, both cards restored, app 0.3.10, the 9070 XT present as gfx1201): the seven final-class packs' 2^24 fingerprints equal on the RTX 5090 (CUDA/NVRTC), the RX 9070 XT (OpenCL) and the M5 Max (Metal): mx8-devnet-epoch0 90f794dd556f7a3b, era-0 8e8e070db4eea52d, era-1 891c01b8563bb47e, era-2 e54279fed2831b5d, era-3 77e0ba8abbd0ae62, era-4 d898d8f4f2e7684b, era-5 a6927db380f7efb2; self-test PASS on every pack on both cards. RULING (coordinator, 5 October 2026, about 21:10 UTC): the integrated gfx1036 (RDNA 2, AMD OpenCL 3683.0) satisfies the AMD vendor tonight, because G1 is a compiler-and-ISA property and gfx1036 carried the v1 and v2 conformance; the 9070 XT's hash-rate and power rows are owed and taken when its link is back | GREEN on the final class (run-ca2-era-pc1b-20261005, 22:16 to 22:21 UTC) |
| G2 | the CPU verifier exact on 1,000 random hashes per card | GREEN: one serve-mode job of 1,024 nonces at target ff..ff per card and pack (every nonce a found line), re-hashed on the Mac with `igneum-pow hash-bound --prehash 00..01 --count 1024` on the same pack: RTX 5090 era-0 1,024 of 1,024 and mx8-devnet-epoch0 1,024 of 1,024; RX 9070 XT era-0 1,024 of 1,024 and mx8-devnet-epoch0 1,024 of 1,024 (the same job) | GREEN |
| G3 | the generator soundness suite green, the new scratch tests included | `cargo test` in igneum-pow, `tests/packs.rs`, the Metal fuzz, edge, stats, determinism runs on the v3 class | GREEN (the crate suite 53 + 4 + 19 + 7 on ca2-mixer 1ab8b21 and the release tree; the Metal fuzz 200 of 200 and 50 of 50 on x8, the edge, stats and determinism runs; the scratch tests 7 of 7; 22:12 UTC) |
| G4 | the fast-time 3-node network mining across a v3 activation | 0 rejected blocks, 0 forks, every node's first lines show the switch, blocks on both sides of the boundary. Run 1 PASS (21:33 to 21:38 UTC, fork 79bd8e10 + igneum-pow 66eeba3, the mixer-x4 class without era): 3 of 3 nodes print the switch line (active from epoch 3, DAA 150 rounded up to 180 at 60-DAA epochs); templates class 2 for epochs 0 to 2 and class 3 for 3 to 5; 181 blocks before and 124 after DAA 180 (305 total, 3 CPU miners); program ids agree on all 3 miners (e3 v3 5d0dedd9fd9e29a1, e4 e81808dcdb02ce05, e5 06aff9c1d33e7a13); rejected 0/0/0 on miners and nodes; one sink 082fd39ba65df2ff on all three at 304/304/304 blocks; a new (day, class) cache 177 to 235 ms on one core, in-day swap 2 ms. `docs/plans/counter-asic-2-node.md` section 5; summary `docs/plans/counter-asic-2-gate/class-v3-20261005-2133Z-mx4.json`. Run 2 PASS (21:46:36 to 21:51:31 UTC, igneumd and igneum-miner rebuilt on b105a55 = the era and mixer composed class, hot None): every check true; 181 blocks before and 124 after DAA 180; v3 program ids e3 2d278041ba482dba, e4 2ae786d294a8a59d, e5 bc36813df2f41b5f on all three miners (the v2 id for epoch 0 8f8806638d59850f unchanged from run 1: v2 byte-identical on the chain too); rejected 0/0/0; one sink 712c1b212091dcdc at 303/303/303; 3 of 3 switch lines; cache ready v2 178 ms, first v3 181 ms, in-day swap 2 ms Run 3 PASS on the FINAL class (22:15:04 to 22:19:44Z, binaries from ca2-v3 d233fa1 = x8 + era + the verifier fix, fork 89dfcb95): 182 / 122 blocks around DAA 180, 304 in all; v3 ids e3 a6523b90cff501e3, e4 bc811b3c4b8b1ced, e5 e784541f19cdebe5 on all three miners; epoch 0's v2 id 8f8806638d59850f the same in all three runs; rejected 0/0/0; one sink a9ce45df8beeaf13 at 303/303/303; 3 of 3 switch lines; cache ready v2 179 ms, first v3 191 ms. Summary `docs/plans/counter-asic-2-gate/class-v3-20261005-2215Z-era-mx8.json` | GREEN (runs 1, 2 and 3; run 3 on the final class) |
| G5 | the PC-built Windows workers and the Mac workers from the same commit | From release-0.3.11 23bc2b2: igneum-worker-cuda.exe 2b3b8c92885442179f6bf2907c6f3eb453dc4a19908d90fd05981a09b7c2674c (1,536,512 bytes), igneum-worker-opencl.exe edc4a75da3b93d814caa69fd635010780d63d5b622ec24c3741d433c584f91e3 (478,208), both with the resource block, different from 0.3.10's pair; the Mac worker and the DMG from the same tree (the shipper's step report) | GREEN at the workers; the DMG and the node builds in flight |
| G4b | the Mac mines v3: a real Metal miner across a v3 boundary through the miner's `--prepare-packs` flow, and the app passes that flag to the Metal worker | Found 22:21 UTC: `app/igneum-app/src/engine.rs` `miner_args` pushes `--prepare-packs` only when `card.worker != "Metal"` (line 1468 on a223ca9), so the Mac app never hands its Metal worker a prepare pack, and under class v3 the Metal worker compiles v3 only from a prepared pack (servePackProgram): at the first v3 epoch every Mac would answer `need` lines and stop, the 18:23Z outage class. Two closes before the ship, both in hand: (a) the app fix on 0.3.11's app branch (push the flag for every worker with the platform's path separator, a unit test on `miner_args` for a Metal card; assigned to the proving agent on top of a223ca9); (b) gate run 4: the fast-time network with one real Metal miner on the Mac across the activation (the prepare lines on both sides, the worker's `prepared` line for the v3 pack, found or accepted blocks on v3, no `need` or mismatch line; assigned to the node agent). If (a) is not in the app tree at the cut, the ship does not go: a Mac that cannot mine v3 at activation is a fleet outage, and the activation height (tip + 14,400) is not far enough to carry the fix in 0.3.12 safely | GREEN. (b) gate 4 (22:25 to 22:31Z, a real Metal miner on node 0, igneum-bench from ca2-v3 00c55aa): three v3 PREPARE lines with the pack dir and class=v3 era=<hex>; the worker's v3 `prepared` lines (252.5 ms the first: program 55.5, dataset 196.9, cache fill 0.9, build 41.1; then 51 and 46 ms with the day resident); every swap "with no pause"; 124 blocks accepted on v3 (301 in the run), cpu re-check mismatched 0, need 0, no mismatch or refusal, no exit 42 or 44; chain 182 / 123 across DAA 180, 0 rejected, one sink. A second outage found and fixed before the run: the Metal worker's serveDataset built every day with the Swift version 2 construction keyed by day only, so a v3 program would have hashed over an x1 dataset; now a v3 prepare builds the day from the pack's memhard.metal and the store keys datasets by (day, class, era), commit 00c55aa |
| G6 | the node change on a fork branch from the 0.3.10 tip 21d4c73c with suites green on PC 2 | the build job id and its SUMMARY line. State 21:27 UTC: fork ca2-v3-node 79bd8e10 (2e464e81 the class switch + ba43cf0f the proving-v1 merge + 79bd8e10 the digest re-pin); Mac: cargo check of the seven crates clean, kaspa-consensus-core 108 + 7, kaspa-pow with igneum-pow 14 (the v3 engine test included); the PC 2 job publishes at 21:45 from the ca2-v3 worktree (the PC's test stage runs kaspa-pow and kaspa-consensus without the igneum-pow feature, so the v3 engine test's evidence is the Mac run). Expected consensus digest for a scratch devnet node with no override file after the flip: c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c (0.3.11; 0.3.10's is 9409deda...) | Mac green. PC 2 job build-20261005-215219 (21:53:01 to 21:55:48Z, 167 s): the Linux build ok, igneum-app tests 78 + 26 + 8 passed, but kaspa-consensus 96 passed and 1 FAILED: processes::finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list, UnexpectedDifficulty(487112384, 487129578) in mine_on_all: the SAME flake the 0.3.10 cut hit on 21d4c73c under the six-package parallel run (release-0.3.10.md section 11: it passes alone, twice). Treated as that cut did: job 2 of 3 build-20261005-215712 (21:57:12 to 21:59:45Z, 122 s): kaspa-consensus alone 97 passed, 0 failed, 3 ignored in 2.10 s, ban_is_decided ... ok; job 3 of 3 build-20261005-220351 (22:04 to 22:06:55Z, 159 s): igneum-exec 17, igneum-miner 18, consensus-core and p2p-flows green, the app 78 + 26 + 8, but kaspa-pow 13 passed and 1 FAILED: igneum::tests::program_class_v3_seeds_hash_their_own_program_over_their_own_cache (consensus/pow/src/igneum.rs:910) compares a v3 program's class to V3_CLASS with era: None, while the merged crate puts the drawn era inside the class (a stale fork test, not a behaviour fault; the PC's stage does run the v3 engine test, so the PC job is the evidence). The fork test is being fixed; job 4 build-20261005-221237 (published 22:12:37Z: main d233fa1 = era + cache + the mixer fix with V3_CLASS = MX8, fork 89dfcb95) ran the five crates and the app on the final tree: every stage ok (22:12:37 to 22:15:44Z): kaspa-consensus-core 108 (2 ignored), igneum-exec 17, kaspa-pow 33 + 14 (the v3 engine test with the igneum-pow feature), igneum-miner 18, kaspa-p2p-flows 7, igneum-app 78 + 26 + 8, 0 failed. With job 2 (kaspa-consensus alone 97) G6 is GREEN on the final tree | GREEN |
If any gate fails: stop at that gate, write why in `docs/plans/counter-asic-2-status.md`, do not publish.
## 8. The release
0.3.11 through the shipper's pipeline (`tools/ship-app.mjs`, the plan shape of `docs/plans/release-0.3.10.md`). The 0.3.10 shipper finishes 0.3.10 first, then takes the 0.3.11 tree, or the coordinator ships 0.3.11 with the same runbook if the shipper has stopped. The publish order is that of `docs/plans/finality-v3-devnet-publish.md`: the override field, the digest handshake, hand nodes and the seed first, then the manifest with `--activation-height` and the deadline note, then the apps, then the digest sweep, then the HiveOS package republish.
### 6c. Card lifetime: cache residency and the growth mapping (from `docs/analysis/card-lifetime-2026-10-05.md`, merged)
| Decision | Choice | The number |
|---|---|---|
| Cache residency on the GPU | DECIDED (5 October 2026, delegated): the cache is FREED after the daily dataset build; the hash reads the dataset and the hot table only, never the cache. hot-table.md's resident reading is corrected to this | The daily rebuild is the only cost: cache fill 0.67 ms and dataset build 13.4 ms on the RTX 5090 at x1 (bench-log, 3 October 2026), 2 ms and 13 to 30 ms on the M5 Max; at mixer x4 the build is about 54 ms (measurement owed on ca2-mixer). Freeing it moves the 12 GB tier from year 12 to year 28 under mapping (b) |
| Dataset growth mapping (a: continuous 2 + 0.5 GiB a year with a multiply-shift index; b: power-of-two steps at years 4, 12, 28, 60 with `AND MASK`) | RECOMMENDED for Josh: (b). It keeps `AND MASK` and every vector's size, it is what option C's "doubles when the dataset doubles" already assumes, and it is the only mapping under which the litepaper's "4 GB about four years, 8 GB more than a decade" is true (under (a) a 4 GB card is out within 1 to 1.5 years, an 8 GB card at 6 to 7.5 years) | card-lifetime table 2: 4 GB out at year 4 (b) or 1.0 to 1.5 (a); 8 GB year 12 or 6.3 to 7.5; 12 GB year 28 freed or 12 resident; 24 GB year 60 freed |
Public lines to fix on the integration branch (card-lifetime table 3): `site/index.html` "2 GB, growing" gains the rate ("2 GB at genesis, doubling at years 4, 12 and 28"); "Any 4 GB card" becomes "any 4 GB card at launch, 8 GB from year 4"; the litepaper's "4 GB about four years, 8 GB more than a decade" stays with mapping (b) and gains "under the step schedule"; `docs/evidence.md` gains a row for the card-lifetime claim labelled designed. hot-table.md line 73's 8 GB row is corrected (it counted a 5090's 8,160 warps; a real 8 GB card has 20 to 24 SMs).
## 7b. Hardware events (for the morning summary)
| When (UTC) | Event | What the app did | For Josh |
|---|---|---|---|
| 5 October, at install (earlier today) | the RX 9070 XT in the Sonnet Breakaway Box 850T5 over USB4 went Code 43 | came back after a driver reinstall and a reboot | |
| 5 October, about 20:40 | the 9070 XT dropped off PC 1's bus: Get-PnpDevice -Class Display lists only the integrated AMD Radeon Graphics (gfx1036) and the RTX 5090; after `pnputil /scan-devices` at 20:45:34Z the card is still absent and the USB4 list shows only the host and root routers: the "USB4 Router (2.0), Sonnet Technologies Breakaway Box 850T5" present at 17:18Z is gone, so the box itself is off the link | the AMD worker (igneum-worker-opencl --device 1) mines the gfx1036 at 3.12 MH/s; the 5090 keeps mining; nobody was woken, PC 1's app was not restarted; a 10-second rescan probe (pnputil /scan-devices, the USB4 router status) was granted | the second eGPU link fault today: reseat the USB4 cable and the eGPU's power; the 0.3.10 hot-plug code shows the card as "removed" and picks it up again without a restart |
| 5 October, 21:22:59 to 22:21 | the card dropped again at 21:22:59 (the third drop), was back and used by the hot-table (21:31 to 21:35), the era (21:46 to 21:51 and 22:16 to 22:21) and the mixer (22:00 to 22:04) jobs as gfx1201, and was gone again by 22:24:09 (the fourth drop; no Sonnet or USB4 router device) | the OpenCL worker falls to the gfx1036 (--device 1) when the card is gone; every measurement that names gfx1201 ran while it was present | the link flaps on a scale of tens of minutes: reseat the USB4 cable and the eGPU's power, try another port or cable; the AMD clock and power sweep is owed on this |
| 5 October, by 21:09 | the 9070 XT is back on PC 1's bus: the reproducible-benchmark package's OpenCL worker listed it as opencl:1 and the app switched it off and on through api/cards; no restart, nobody touched the box | the app's AMD worker returns to it at the next prepare | the link drops and returns by itself; the reseat is still worth doing in the morning |
| 5 October, by 21:22:59 | the third drop: no Sonnet or USB4 Router (2.0) device present, the display list shows only the gfx1036 and the 5090 | the app lists nvidia:0 and amd:1:gfx1036; the 5090 keeps mining at 124.5 MH/s (app log STATUS lines through 21:23:26Z) | the link is flapping: reseat the USB4 cable and the eGPU's power in the morning, and consider a different USB4 port or cable; every AMD measurement tonight runs on the gfx1036 fallback unless the card is present at the job's own probe |
## 7a. A dated constraint from the consequences review (C1)
The fee switch H = 210,000 arrives about 18:45Z on 6 October (DAA 137,041 at 22:32:54Z on 5 October; 1.002 DAA/s averaged since 15:40Z; the 19:50Z in fee-switch-devnet.md is an hour late). The 0.3.10 node's export RPC carries no daaScore and no feesV1ActivationDaa (the proving v1 fork does), so from H every app prover on 0.3.10 has its shards refused and the devnet's proving goes dark. 0.3.11 must be on every prover before 16:00Z on 6 October; if it is not, the fee switch is republished at H = tip + 86,400 by the fee-switch plan's rule (a digest flip, every node in one sweep). The status file carries the timing against H.
## 8a. Proving v1 rides with it
Josh delegated the proving v1 decisions to its agent (acd4f36bc2c07a4e2). Handoff received 21:05 UTC: fork proving-v1 = ece42979 on 21d4c73c (eb32c645 the feature on protocol 15 / message 75; 2dfad910 the rebased pool test; 3203c8d0 N = 8; ece42979 the digest test edit); Mac unit tests on it: consensus-core 26, exec 8, flows, 0 failed; app proving-v1 FINAL e0de2ab on 5b0d54f (a223ca9 the resume fix, e0de2ab the Metal --prepare-packs fix; docs-only commits between) (6dc686a plus the resume fix: the resume path re-arms every slot without a live worker, re-exports the pack, and logs "<card> is not mining 90 s after resume"; the known-failed case of PC 2's 21:25:11Z resume is the unit test the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old, `cargo test -p igneum-app resume` 3 passed; cause: Cmd::Resume re-armed only faulted slots after stop_miners had cleared every restart_at; the tier numbers from the miner-on curve, the AMD and Apple "mines and does not prove" line, the CPU path refused under 32 GB of RAM, the 24 GB tier marked "measured on the 32 GB card", the fast-time file at unproven_daa 10; `cargo test -p igneum-app provedefault` 6 of 6 on the Mac). Harness on the final fork tree: `IGNEUM_PV1_BIN=vendor/igneum-node/target-pv1/release tools/lock/with-lock.sh run node tools/proving-v1/net.mjs --secs 1500 --segment 8` gave "RESULT proving v1 harness: PASSED (21 checks) in 244.4 s" at 20:56:45Z. Override fields at publish: proving_v1_activation_daa = tip + 14,400, proving_v1_segment_blocks 8, proving_v1_unproven_daa 600, proving_v1_aggregator_share_bps 1000; they enter the digest only once the activation is set. Pinned guests unchanged. Mixed fleet: a 0.3.10 node peers with a 0.3.11 node at protocol 14 and never receives message 75; it carries segment records as miner bytes and pays nothing for them; before H the fleet is unchanged, after H only 0.3.11 producers carry and pay segment records and the shard split moves to 90/10, so every node must be on 0.3.11 before H (the fee-switch rule, section 7a). The PC 2 suite job on the merged tree (ca2-v3-node + ece42979) is the suite evidence for both halves. The app half also carries the resume fix (assigned 21:47 UTC to the proving agent on its app branch on top of 6dc686a): engine.rs's resume path restarts every enabled card's worker and the pack export if the pack is stale, re-checks within one tick that every enabled card is mining and logs a failure naming the card if not, with a unit test on the state machine (paused -> resumed -> every enabled card mining within one tick) and the known-failed case of PC 2's 21:25:11Z log; the defect left PC 2's miners off for 20 minutes tonight and the Mac's miner off after pause+resume this afternoon. If its commit is not in hand when the app tree is cut, it is first on 0.3.12's list and the status file says so. 0.3.11 carries program_class_v3 AND proving v1 together: one override object, one digest, one publish, the same gates for each half (its fast-time harness green on the final tree, suites on PC 2); both activation heights set at publish by the same rule (tip + 14,400, checked >= 10,800). If one half is not ready when the other is, the ready half ships as 0.3.11 and the other as 0.3.12; the status file says which.
Merge rule for the ship (C31): the app branch proving-v1 (e0de2ab) rewrote docs/evidence.md row 16 (WITHDRAWN, the 24 GB measurement) and the litepaper's proving sentences; ca2-coord carries the older row 16 and its own litepaper edits, so at the merge take proving-v1's row 16 and its proving sentences, and ca2-coord's everything else; the stale row must not win by accident. fud-close (the ledger closer's main-repo branch: 45 public-text fixes on the site and litepaper, spec 8.3 and 8.8, two CI checks, the relay fixes; a merge-tree onto ca2-coord shows 0 conflicts) is NOT in 0.3.11: the tree closed at 23bc2b2 (its workers, DMG and PC build job carry it) before the branch reached the ship order, and the shipper takes no late branch (the 0.3.10 rule); fud-close heads the next cut's list, with a coupling the next cut must respect: fud-close's worker change for ledger M28 (packfile.h's kernel_sha256 check, host.c refusing a pack that fails it) pairs with the fork-side miner change on ledger-fixes 3d4ec451 that stamps the hashes into program.json; the workers without that miner commit refuse every pack, the miner without the workers is harmless, so both go in one cut or the worker half of M28 is held back. fud-close tip 647b08c (its checks green); ledger-fixes is not yet rebased onto 89dfcb95 (two conflicting files: igneum/miner/src/main.rs, protocol/flows/src/ibd/proof.rs). The fork-side ledger-fixes branch (from release-0.3.6, not 21d4c73c) is NOT in 0.3.11: it rebases onto the 0.3.11 fork for the next cut. Branches merged into the 0.3.11 main tree beside ca2-v3 and ca2-coord (tooling, no consensus): consequences (the reviewer's rows), bash-body-check (7adb1ca, 6805125: tools/ci/bash-body-check.sh with fixtures and a self-test in ci.yml, the convention paragraph in packaging/README-ship.md, `publish-jobs.sh add --kind run` running the bash-body and prover-socket checks before signing; add/add with proving-v1 c2544be on tools/ci/prover-socket-check.sh: take bash-body-check's superset and proving-v1's ci.yml step; tools/amd-prove/pc1-cpu-prove.ps1 goes on the allow list, it starts no GPU server), amd-prove (f1d7a7d), card-lifetime (1fecfe2). Next-cut list (not consensus, not in 0.3.11 unless a one-file app change with tests): ota-k2 (branch ota-k2, commit c722579e, "OTA: the second signing key (K2) with revocation": manifest.rs, ota.rs, jobrun.rs, jobs.rs, inputs.rs, ota-sign.rs, engine.rs, the publish scripts, tools/keys, docs/security/keys.md section 4; ships signed with K1; K2's public half is empty until Josh runs tools/keys/keygen-k2.sh), ember-tune 8ab9068 (the second-engine rule: no pipe into a second engine, its process tree killed at the end and on the budget, the installed app's miners restarted after; applied to ember-tune-pc1.ps1 and sweep-5090.ps1; tools/ci/second-engine-check.sh in ci.yml, shown to fire on a known-bad playbook and pass the fixed pair; every Cmd::Quit stamped with its source; the AMD gmax offset fix); rig-install (branch rig-install dd632c1, done) with two follow-ups that belong with 0.3.11 if the proving half ships then (a rig that cannot prove defeats the point), else 0.3.12: (i) the signed public manifest names no Linux package, so the installer verifies the HiveOS tarball through the unsigned downloads sidecar behind a flag; fix = `publish-public.sh --hive` adds a `platforms.linux` entry and re-signs (the apps ignore the extra key); (ii) the published Linux package carries no prover binaries and no key-hash / sign-record miner, so the rig's prover unit idles in "setup"; fix = a Linux prover build (sp1 host, pinned guests) in the cross-build set and the package; pool-v0 (its own service, no app change), repro-bench; per-day dataset reuse in the CUDA and OpenCL workers (a Day object split out of Pair: cache freed after the build, dataset kept across prepares of the same day; the Metal worker already does this; first item after the publish, for 0.3.12; until then the integrated tier mines v3 with a restart per epoch); the rig miners' `--exit-on-seed-change` path (exit 42, re-export on restart) replaced by prepare-ahead before any epoch shorter than an hour can be drawn (layer 9 precondition, consequences C20); fork-side pack-loop 05ef0fa3 is merged into the v3 node branch because v3 touches the same miner paths.
## 9. After the publish: Counter ASIC 3.0
The ASIC-history agent sends its ranked additions; they are measured the same way, folded into the class as v4 behind its own activation, same gates, same rollout; `docs/plans/counter-asic-3.md`.
## 10. Decisions table (superseded by section 6; kept for the options)
| Decision | Options | Recommendation | Source |
|---|---|---|---|
| The width rule (layers 1 and 2) | 4 B fixed; 16 B; 64 B; the per-program mix | `<the readwidth table's decision rule: the widest read that keeps every card latency-bound with margin on the 5090>` | `docs/plans/read-width.md` |
| The scratch share and size (layer 3) | 0, 12.5, 25, 50% RMW at 32 or 128 KB per warp | `<from the readwidth table and docs/analysis/scratch-soundness.md>` | the same |
| The hot table (layer 5) | 32, 64, 96 MB; replaced or added | DECIDED (5 October 2026, 21:36 UTC, delegated): OUT of v3. The rule was g at or above 0.97 on both PC cards in the added form; measured g is 0.87 / 0.85 / 0.84 on the 5090 and 0.84 / 0.81 / 0.80 on the 9070 XT at 32 / 64 / 96 MiB: neither card keeps even the 32 MiB table resident while the dataset streams, and the replaced form helps the on-die-cache chip. Layer 5 stays a measured option for 3.0 | `docs/plans/hot-table.md` (ca2-cache): 5090 MH/s v2 136.1; replaced hot32k4 146.6, hot64k4 140.8, hot96k4 138.5, hot64k2 137.5, hot64k8 163.6; added hot32k4a 118.7, hot64k4a 115.4, hot96k4a 114.4. 9070 XT v2 18.15; replaced 19.79 / 18.73 / 18.33 / 18.17 / 22.32; added 15.27 / 14.62 / 14.56. M5 Max added 0.93 / 0.87 / 0.83. All eight packs bit-exact on the 5090 and the 9070 XT with the Mac's fingerprints (21:29 to 21:35 UTC, no restart straddled) |
| The era draws (layers 4 and 8) | in, if the min-to-max spread across six drawn eras is under 5% per card; item size fixed at 4 B, draws of stride, interleave and the working-set window at or above 256 MiB | DECIDED (5 October 2026, 21:58 UTC, delegated): IN. Spread over six eras: RTX 5090 1.3% (136.18 / 136.44 / 138.01 MH/s against v2 137.2), RX 9070 XT 3.2% (18.61 / 18.93 / 19.21 against 18.09), M5 Max 0.8% (28.35 / 28.48 / 28.58 against 27.68); all under 5%; latency-bound share 1.01 / 0.95 / 1.06; every pack's 2^24 fingerprint equal on all three vendors; the CPU verifier 1.00 to 1.02 of v2 within one binary | `docs/plans/era-layout.md` (ca2-era 78c0ee4; PC 1 jobs fetch-ca2-era-20261005 and run-ca2-era-pc1-20261005, 301 s, both cards restored). Chip line: 512 B read per hash, 120 to 128 distinct lines, the mirror is the whole dataset every hour; the interleave's value against a chip with a programmable address decoder is nil (stated), the stride is a bijection with no cryptanalysis yet |
| The cache schedule (layer 6) | flat 256 MiB or a growth schedule | DECIDED (5 October 2026, delegated under "execute the full 2.0 plan, tonight 1-8"; Josh confirms for the public testnet genesis; the mirror's mm^2 are being corrected to shipped-product density, 2 to 3x the bit-cell figures, conclusion unchanged): option C, the cache doubles when the dataset doubles: 256 MiB at genesis, 512 MiB at year 4, 1 GiB at year 12. Verifier fill on one M5 Max core at 0.2 s per 256 MiB (spec 1.12): 0.2 s, 0.4 s, 0.8 s at each step, under 1 s at every step of the schedule; verifier memory 256 MiB, 512 MiB, 1 GiB | `docs/analysis/sram-mirror.md`: a 256 MiB mirror is 54 mm^2 at N2 (0.0175 um^2 cell, array factor 0.70), about $26 per good die, approximate |
| Layer 7 | reserved family, unlock by era height or 90% signal | DECIDED (5 October 2026, delegated): reserve family R1 = mm8 (uint8 8x16 by 16x8 tile per unit, unsigned bytes), W_new 4, unlock at era 4 or 90% signal, the emulation rule in spec 1.13.2; switched off, no consensus effect tonight; the 5090 and 9070 XT dp4a numbers when the PCs free (PC 2 first) | `docs/analysis/int8-matrix-family.md`: native on PTX mma.sync, AMD WMMA iu8, Metal 4 matmul2d; dot4 emulation on Apple 1.6x (unsigned) |
| The activation height N4 | the rule of section 3 | N4 = N5 = 154,800 (the shipper, 22:45 UTC: DAA 136,967 at 0.965 blocks/s puts the publish near 140,200, tip + 14,400 near 154,600, the first multiple of 3,600 at or above is 154,800; one number for both switches); the 10,800 floor holds until DAA 144,000, about 00:45 UTC, re-pinned and rebuilt past that | release-0.3.11 23bc2b2 (packaging/mac/packaged-config.sh, the nine-field line) |

View file

@ -0,0 +1,633 @@
# Counter ASIC 2.0: status
Coordinator's running status for the plan in `docs/plans/counter-asic-2.md`. Rewritten every 45 minutes while the work runs. Times UTC, 5 October 2026 (night). Heading times before 20:40 were corrected at 20:42 from the commit clock (the coordinator had written them from a guessed clock, up to 2 h 40 min ahead; corrected again at 20:49 for the 20:42 to 20:47 entries); every entry's true time is its commit's author time in UTC, and from 20:49 every heading is stamped from `date -u`. Base for every ca2 branch: `readwidth` at 019b014 (the LoadClass flag, fold_words, the scratch op, the three emitters, 20 packs).
## 19:55 first status (the 19:50 start was cut off by exhausted credits at about 19:58 before any sub-agent work landed; respawned at 19:55 on the restart)
| Layer | Branch | Agent | State | Numbers so far | Blockers |
|---|---|---|---|---|---|
| 1, 2, 3 (widths, mix, scratch) | readwidth 019b014 | a451c9935bfb1bc19 (not ours) | Mac Metal rows in; Apple OpenCL next, then one job per PC; scratch re-run at 32 and 128 KB per warp | CPU verifier, M5 Max one core, avg of 50, ms per warp: v2 0.604, w16 0.610, w64 0.630, w64x4 0.160, mixA 0.620, mixB 0.614, scr0 0.600, scr2 0.534, scr4 0.457, scr8 0.314. Metal M5 Max hash rate (5 x 2^24, GPU time, 3 of 3 vectors on every pack): v2 27.7 MH/s, w16 28.3, w64 28.2, w64x4 109.7 (32 loads of 64 B), mix 50/35/15 over 6 programs 25.4 to 28.4 (median 26.7), mix 25/50/25 over 6 programs 22.0 to 25.2 (median 24.4). Scratch at 1 MiB per warp, 1,024 warps: 23.1 / 18.2 / 16.3 / 14.9 MH/s at 0 / 12.5 / 25 / 50% RMW, superseded by the cap | holds the Mac measure lock and both PCs first |
| 4 + 8 (era layout, working set) | ca2-era | a452664c512c73b9b | design, respawned 19:56 | none | PC time behind readwidth |
| 5 (cache-sized second table) | ca2-cache | a5271cf269757b118 | design, respawned 19:57 | none | PC time behind readwidth |
| 6 (SRAM schedule) | ca2-analysis | a5c6bc2dfcc4613ef (respawned 19:58) | cited analysis | none | none |
| 7 (integer matrix family) | ca2-analysis | a5c6bc2dfcc4613ef | design; dp4a throughput owed unless PC time frees | none | Apple int8 path to check against Metal docs |
| 3 soundness | ca2-soundness | a548aadeefd1ab3b2 (spawned 19:59; sizes 32 and 128 KB per warp) | analysis + tests | none | none |
| Integration v3 | ca2-v3 | after the above | waiting | none | the readwidth table and the four branches |
Budget rule received from the coordinator at the restart: the per-warp scratch is capped so the whole working set (1 GiB table + hot table + scratch for every resident warp + buffers) stays under 6 GB on an 8 GB card, which puts the scratch in the tens of KB per warp; the layer 5 hot table shares that budget.
Decisions for Josh so far: none. Nothing here touches consensus or any live node.
## 20:05 readwidth PC jobs out; the scratch finding
Base moved: every ca2 branch rebases onto readwidth b970dda (scratch per warp is a class parameter, 32 or 128 KiB; the SCRATCH_* constants are gone; Metal pack harness `proto-metal/packbench.swift`; OpenCL `--bench-pack`). The coordinator branch is rebased; the four agents were told.
Readwidth PC jobs published 20:02:51Z: `fetch-readwidth-20261005` (both PCs), `run-readwidth-5090-20261005` on PC 2 (6 to 10 min once it starts; a build job from another session is queued ahead of it), `run-readwidth-9070-20261005` on PC 1 (10 to 15 min; only the gfx1201 card is switched off). My layer 4, 5 and 7 PC jobs queue behind these.
Finding that bears on layer 3 (readwidth, Metal, M5 Max, capped sizes): the scratch read-modify-writes are cheaper than the dataset loads they replace, so the rate RISES with the RMW share.
| Class | MH/s |
|---|---|
| v2 | 27.7 |
| scr0k32 (control) | 28.1 |
| 32 KB per warp, 12.5 / 25 / 50% RMW | 25.4-26.1 / 29.4-31.7 / 44.4-49.1 |
| 128 KB per warp, 12.5 / 25 / 50% RMW | 24.3-26.2 / 26.4-28.1 / 34.1-35.4 |
Reading (readwidth agent): 4,096 warps x 32 KB = 128 MB sits in the chip's caches. Consequence for the decision: a scratch that fits a GPU's cache fits a chip's SRAM at the same size, so at the capped size the writes cost everyone a cache-bound op in place of a latency-bound load. Passed to the soundness agent: what size would make the writes cost DRAM latency, whether that fits the 6 GB cap, and whether the RMW share should be added to the 16 dataset loads rather than taken from them.
## 20:08 mandate: v3 on the devnet tonight, by Josh's rules
Josh has gone to bed and delegated the three decisions for the DEVNET only (not the public testnet): width, per-load mix and scratch share by the rules now written in `docs/plans/counter-asic-2-rollout.md` section 6, the activation height = devnet tip + 14,400 at publish (checked >= 10,800), published the way finality v3 was. Six gates before any publish (rollout section 7): bit-exact v3 on the three cards; the CPU verifier exact on 1,000 random GPU hashes per card; the soundness suite green with the new scratch tests; the fast-time 3-node network mining across a v3 activation with 0 rejected blocks and 0 forks; Windows and Mac workers from one commit; the node change on a fork from the 0.3.10 tip (21d4c73c) with suites green on PC 2. Release 0.3.11 through the shipper's pipeline; the 0.3.10 shipper (ae892a8b0f78fe31c) has been asked for its state and the handoff. If a gate fails: stop, write why here, do not publish.
Node build note for the integration: the node links `igneum-pow` by path (`../../../../igneum-pow` from `consensus/pow` and `igneum/miner`), so the node fork worktree for v3 must live under the v3 worktree's `vendor/` so that the path resolves to the v3 crate, not master's.
After the publish: Counter ASIC 3.0 from the ASIC-history agent's ranked additions (a202a09dcd24ba1d3), as class v4 behind its own activation, same gates, `docs/plans/counter-asic-3.md`.
## 20:08 the shipper's answer, the node fork convention
0.3.10 (shipper ae892a8b0f78fe31c): staged and blocked on GitHub's Actions incident (run 37365130137 queued since 19:42:43Z under a re-dispatching watcher); nothing on the network has moved, the live manifest is still 0.3.9. Once CI is green: ship (5 min), update-now (apps restart 1 to 10 min later), hand nodes and seed (5 min), digest sweep; 0.3.10 finished about 30 min after green. The app version per machine on the console's cards is the restart signal for re-running any straddling measurement. HiveOS is published by the ship's --public step, nothing separate.
Node fork for v3: base on COMMIT 21d4c73c (release-0.3.10 in vendor/igneum-node = release-0.3.6 a24ab01a + housekeeping 4fb32865 + tx-gossip e242acd0 + c4-fix e18f1e0e). Fork-side pack-loop 05ef0fa3 (the miner's pack check, exit 44) is not in it and touches the same miner paths as v3 (export-pack, the job line): merge it. Because the node links igneum-pow by relative path, the v3 node worktree goes under the v3 worktree: `git -C /Users/joshm/Projects/igneum/vendor/igneum-node worktree add /Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2 -b ca2-v3-node 21d4c73c`.
0.3.11 shipper: the coordinator assigns it (the 0.3.10 shipper stops at its report). Inputs the ship needs: the fork commit with its PC 2 suite results recorded, the main tip, the override object with every switch (the four live fields plus program_class_v3_activation_daa), the expected digest read on a 22-s scratch node, the deadline note ("program class v3"), the activation height, the one-line changelog, and whether the pinned proving guest changes (it does not: the prover has no igneum-pow dependency; confirmed by grep of proving/igneum-prove Cargo files).
## 20:10 readwidth round 2 on the PCs; the width arithmetic under Josh's rule
Round 1 of the readwidth PC jobs refused every pack (the workers demand a 32-byte chain seed; the experiment packs carried string seeds); fixed at readwidth 1ea7a52 (packfile.h), republished 20:09:19Z as `run-readwidth-5090-20261005c` (PC 2, about 6 min) and `run-readwidth-9070-20261005c` (PC 1, about 10 min). The era and cache agents were told to rebase onto 1ea7a52 and to prove their packs load on the Mac OpenCL host before any PC job. Every ca2 branch now bases on 1ea7a52.
Probe ceilings from round 1 (dependent reads per second at 1024 MiB, device time):
| Card | 4 B | 16 B | 64 B | Stream |
|---|---|---|---|---|
| RTX 5090 | 17.5 G | 18.0 G | 9.1 G (584 GB/s) | 1,579 GB/s |
| RX 9070 XT | 2.42 G | 2.43 G | 2.47 G (158 GB/s) | 636 GB/s |
Josh's width rule applied to the ceilings alone (the measured v3 rates will replace this when the table lands): a 128-load hash at 64 B reads 8,192 B; at the 5090's 64 B ceiling that is 71 MH/s and 584 GB/s, 37% of its stream bandwidth, over the one-third margin the rule sets; at 16 B it is 2,048 B per hash, 141 MH/s and 288 GB/s, 18%, inside the margin; 4 B is 9%. On the 9070 XT every width costs the same line fetch (2.4 G/s), so 16 B is where AMD gains 4x the bytes per hash at no cost and the 5090 stays latency-bound with margin. Provisional width under the rule: 16 B (w16), pending the measured rates and the latency-bound share per card.
Added deliverable (20:11): the public description in four levels, `docs/plans/counter-asic-2-public.md` (aec53bb): levels 1 and 2 are written as copy; level 3 carries the bench table with `[owed]` markers for every number not yet measured; level 4 lists the documents. The integration branch applies levels 1 to 3 to `site/index.html`, `site/litepaper.html` and `site/bench.html` with the final numbers.
## 20:15 layers 6 and 7 landed; the node-fork agent started; 0.3.11 scope
Branch ca2-analysis (5d5ba15, f59708d). Layer 6: no cache growth rule exists in the spec; cited bit cells N7 0.027, N5 / N3E / Intel 18A 0.021, N3B 0.0199, N2 0.0175 um^2, array factor 0.70; the 256 MiB mirror is 83 / 64 / 54 mm^2 at N7 / N5 / N2 (74 at N2 with a 96 MB hot table), $13 to $26 of silicon per good die (approximate). The mirror was never unaffordable; the cache's job is to stay above GPU L2 (5090 96 MB, GB202 128 MB). Recommendation C for Josh (gate 1): cache doubles when the dataset doubles. Layer 7: Metal 4 matmul2d has int8 x int8 -> int32, so a unit-level mm8 tile is native on all three vendors; per-lane dot4 is emulation on Apple (M5 Max: 548 G unsigned dot4/s against an 880 G ALU chain, 1.6x; signed 4.7x). Reserve R1 = mm8, W_new 4, unlock era 4 or 90% signal. Owed: the 5090 and 9070 XT dot4 probe (job prepared: relay/playbooks/dot4-probe.ps1, exe sha256 5adaeb1a...6416f4; publishes when a PC frees).
Node-fork agent a3f505a9d981300cd started 20:50: ca2-v3 (igneum-pow seam: ProgramClass, Epoch::from_chain_seeds, generator 3 in the program id, class in the pack) and ca2-v3-node (vendor/igneum-node-ca2 under the ca2-v3 worktree, from 21d4c73c, with pack-loop 05ef0fa3 merged): the field in Params, OverrideParams and the digest, the epoch-boundary rounding, the era stand-in, the job line, the fast-time gate script.
0.3.11 scope (coordinator, 20:52): carries program_class_v3 and proving v1 together (one override object, one digest, one publish; each half under the same gates); the ready half ships as 0.3.11 and the other as 0.3.12 if one lags. Next-cut list recorded in the rollout plan section 8a.
## 20:16 decisions recorded: layer 6 option C, layer 7 R1 = mm8; the chip headline
Layer 6 DECIDED (delegated; Josh confirms for the public testnet genesis): option C, the cache doubles when the dataset doubles (256 MiB genesis, 512 MiB year 4, 1 GiB year 12); one-core fill 0.2 / 0.4 / 0.8 s, under 1 s at every step. Layer 7 DECIDED: reserve R1 = mm8, unsigned, W_new 4, unlock era 4 or 90% signal. The chip model's headline now names the on-die-cache recompute chip (54 to 83 mm^2) as a row per variant; the scratch share is chosen as the smallest share at which that chip's gain falls under 1.5x, else said plainly and the public "under 2x" claim qualified. The soundness agent carries that table (its question 2) with M16's mixer multiplier beside it.
## 20:18 correction to the SRAM mirror figures (chip-economics research, cluster D)
Bit cell x 0.70 understates real die area. Shipped cache dies: AMD 3D V-Cache 64 MB on 41 mm^2 at 7 nm (1.56 MB/mm^2, Tom's Hardware, Hot Chips August 2021); Graphcore GC200 900 MB on 823 mm^2 with compute (1.09 MB/mm^2); Groq TSP 220 MB on 725 mm^2 at 14 nm (0.30 MB/mm^2); TSMC N5 HD SRAM macro 31.8 Mib/mm^2 after about 30% assist overhead (SemiAnalysis, December 2022). A 256 MiB mirror is about 165 mm^2 at 7 nm on the densest shipped cache-only die and about 130 mm^2 at N5/N3E, not 54 to 83 mm^2; cost per die 2 to 3x the earlier figure; the conclusion (affordable for a funded chip) stands. The analysis agent is redoing the table with both columns; the soundness agent carries the corrected density into the chip row. Latency citations behind the latency-bound rule, to be added: DRAM row cycle 40 to 48 ns across DDR4, GDDR5, HBM2 (Li, Reddy, Jacob, MEMSYS 2018); latency 1.3x in two decades against bandwidth 20x (Chang 2017); no shipped mining chip used HBM or stacked memory.
## 20:19 proving v1 state for 0.3.11; PC 2 occupancy
Proving v1 (acd4f36bc2c07a4e2): fork proving-v1 b177718e on a24ab01a (told to rebase onto commit 21d4c73c now), app proving-v1 79bc820 on a93199a. Override fields proving_v1_activation_daa (tip + 14,400 at publish), proving_v1_segment_blocks 4, proving_v1_unproven_daa 600, proving_v1_aggregator_share_bps 1000; they enter the digest only once the activation is set. Harness: `tools/proving-v1/net.mjs --secs 1500` PASSED (21 checks) in 197.3 s on b177718e; rerun owed on the final tree. The pinned guests do not change. Shared files with ca2-v3-node: params.rs, daemon.rs, igneum/miner/src/main.rs, override-60x.json; both agents keep separable hunks.
PC 2 is held by the proving agent's memsweep-pc2-pv1 (about 20 min, miners stopped) and a second run (about 10 min). Queue after it: the ca2 node suites, then the readwidth, era, cache and dot4 measurement jobs. PC 1 is held by readwidth's run-readwidth-9070-20261005c until it reports.
## 20:21 sram-mirror.md revision 2 (ca2-analysis)
Two columns, headline = shipped-product density (AMD V-Cache 41 mm^2 per 64 MiB at N7, scaled by the bit-cell ratio), lower bound = bit cell x 0.70. mm^2 and $ per good die (D0 0.1 per cm^2, wafer prices approximate), headline / lower bound:
| Node, wafer $ | 256 MiB | 256 + 96 MB hot table | 1 GiB |
|---|---|---|---|
| N7, $9,500 | 164 / 83 mm^2, $30 / $13 | 226 / 114, $44 / $19 | 656 / 331, $224 / $75 |
| N5, N3E, 18A, $20,000 | 128 / 64, $46 / $21 | 175 / 89, $68 / $30 | 510 / 258, $306 / $111 |
| N3B, $20,000 | 121 / 61, $43 / $20 | 166 / 84, $63 / $28 | 483 / 244, $280 / $103 |
| N2, $30,000 | 106 / 54, $56 / $26 | 146 / 74, $81 / $37 | 425 / 215, $343 / $131 |
Year 10 at the 6% per year trend: 59 mm^2 for the flat cache (82 with the hot table), 8% of a 750 mm^2 die. One reticle holds 1.3 GiB (N7) to 1.9 GiB (N2); mirror share of a 750 mm^2 die at year 0: 14% (22% with the hot table), inside M16's 13 to 40% band. Recommendation unchanged: C. Latency section added (MEMSYS 2018, Chang 2017, the mining-chip memory-type note), marked as research the agent did not re-read tonight apart from the V-Cache figure.
## 20:23 layer 3 soundness landed: the scratch does not move the chip; scratch share decided 0
ca2-soundness (0d8f745 tests and trace hook, a465881 doc and bench-log). The on-die-cache recompute chip (N5 headline 128 mm^2, $46) at 50 T op/s: 333 MH/s against the 5090's measured 139.7, 2.4x; at 12.5 / 25 / 50% RMW replaced, chip 381 / 443 / 661 against 5090 projected 160 / 186 / 279, 2.4x each; added, 2.4x or more; 32 or 128 KB alike. The chip keeps the scratch implicitly in 80 to 320 B per lane (the verifier resets it per unit), needs about 530 units in flight, dense scratch 6.2 / 3.1 mm^2 at N5. Under Josh's rule the scratch share is 0: layer 3 is NOT adopted into v3; the public "under 2x" claim is qualified (public copy level 3 rewritten). The measured lever is M16's mixer multiplier (x2 1.2x at 0.8 to 2.4 ms verify; x4 0.6x at 1.6 to 4.8 ms; 3.6x and 1.8x with a 3x fixed-function factor); whether x4 enters v3 tonight is asked of the coordinator; default: Counter ASIC 3.0.
Soundness results (Metal, M5 Max): 28/28 edge launches, 200/200 fuzz packs (91 s), 56/56 hand-model edge checks, 42/42 kernels pass the static scratch-mask check with 6 deliberate breaks caught, broken tag and broken lazy fill caught, fingerprint 8c07620f4d9adefd warp-count-independent; re-hit rates 2 to 33% above the birthday bound (slot addresses are register low bits); written words unbiased (worst 3.63 of 6 sigma). Verifier exactness needs a host contract (zero the arena at allocation and at the 32-bit tag wrap, tags from 1), which neither host gives today. Pre-existing on readwidth b970dda: verify::tests::fold_and_wide_fetch overflows under the test profile (wrapping_mul fixes it); passed to the readwidth agent with the class sweep.
Gate G3 note: the scratch tests (igneum-pow/tests/scratch.rs) join the v3 suite even though the class carries no scratch, parametric over the class; they guard the v2 path's scratch-free invariant at zero cost.
## 20:24 decided: M16 mixer x4 into v3; agent ca2-mixer started
Coordinator's decision under Josh's delegation (recorded in the rollout plan section 6a): the mixer multiplier x4 and the cache growth rule (option C) enter class v3 behind the same activation; layer 3 stays out at scratch share 0, its soundness document and pack-contract tests kept. Agent af345b1e2c541ffbb (branch ca2-mixer) implements `mixer_mult` as a class parameter (m mixer applications per round, the 8 dependent reads unchanged), the `cache_log2_words(day)` schedule (doublings at years 4 and 12 with the dataset stepping to the next power of two), re-cuts the v3 dataset vectors, re-runs the soundness suite, measures the verifier (v2 0.604 ms per warp; v3 expected 1.6 to 4.8 ms) and the 1 GiB build on the Mac, prepares the 5090 and 9070 XT build-time job, and writes docs/analysis/chip-model-v3.md with the combined headline row (fixed-function factor included). The claim on the site reads "under 2x" only if that row does; else qualified, with the mixer x8 and the hot table named as the next levers.
Agents now: ca2-era (a452664c512c73b9b), ca2-cache (a5271cf269757b118), ca2-node (a3f505a9d981300cd), ca2-mixer (af345b1e2c541ffbb). Done: ca2-analysis, ca2-soundness. Waiting: the readwidth PC table; PC 2 (proving memsweep runs) and PC 1 (readwidth 9070 round).
## 20:27 the readwidth table landed; layers 1, 2, 3 decided; layer 5 measured on the Mac and redesigned
Readwidth e752fc7 (`docs/plans/read-width.md`), bit-exact on Metal, Apple OpenCL, the 5090 (NVRTC) and the 9070 XT, both PCs released. MH/s (latency-bound share):
| Class | RTX 5090 | RX 9070 XT | M5 Max | Gap |
|---|---|---|---|---|
| v2 (128 x 4 B) | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x |
| w16 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x |
| w64 | 71.9 (0.58, 37% of stream) | 17.59 (0.78) | 28.27 (1.03) | 4.1x |
| w64x4 (32 loads) | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x |
| mix 50/35/15, six programs, spread of median | 18.8% | 7.4% | 11.3% | |
| mix 25/50/25 | 22.3% | 5.5% | 8.1% | |
| scratch 32 KB at 12.5 / 25 / 50% | -18 / -21 / -12% | -18 / -22 / -21% | -7 / +12 / +74% | |
| scratch 128 KB | -21 / -30 / -48% | -21 / -27 / -33% | -7 / -1 / +25% | |
Decisions (Josh's rules, delegated): layer 1 keep v2 (w16 passes the rule but closes nothing and does not move the chip row; the vector re-cut is not worth it); layer 2 out (spread over 5% on every card); layer 3 out (scratch share 0). The AMD gap is the card's dependent-read rate (2.4 G/s at every width), stated for the user tiers in the rollout plan 6b.
Layer 5 (ca2-cache 53ef59f, 011cc0a, 86726cd, 65bc7a7; `docs/plans/hot-table.md`): five packs bit-exact on Metal and Apple OpenCL (96/96 each). M5 Max rates against v2 27.68: hot32k4 x1.22, hot64k4 x1.12, hot96k4 x1.05, hot64k2 x1.00, hot64k8 x1.71; verifier 0.344 to 0.560 ms against 0.626; hot fill per epoch 24 / 46 / 73 ms on one core, 0.07 / 0.15 / 0.22 ms on the GPU; Apple OpenCL probe 32 / 64 / 96 / 1024 MiB 21.7 / 12.8 / 12.3 / 3.50 G loads/s. Redesign ordered: hot loads ADDED beside the 16 dataset loads (the replaced form lets the on-die-cache chip skip item derivations and worsens the gain); the agent re-measures the added form and rebuilds the PC job. PC 1 is given to the dot4 probe (under 15 min), then to the era agent, then the hot-table job; PC 2 stays the proving agent's.
## 20:27 layer 9 added; the era draw passes Mac bit-exactness; two class bugs in the harnesses; C1
Layer 9 (Josh: faster program changes): the epoch length becomes an era parameter in the genesis reserve, 1 hour at launch, 10 minutes to 2 hours by draw or 90% signal, reserve-only tonight; an agent (ca2-epoch) designs it beside layers 4 and 8 and measures the compile-ahead cost per card at a 10-minute epoch, the seed-path consequence and the FPGA threat it answers; one row in the rollout plan section 6, one in the level 3 numbers. Spawns when the dot4 probe frees its slot.
ca2-era mid-way (a452664c512c73b9b): era draw behind LoadClass::era (EraParams beside mix, load_slots, scratch, scratch_kb; verify::load_index; memhard::Layout; one load form in the three emitters; --era, --era-widths), 54 crate tests green, pinned packs byte-identical; six era packs bit-exact on Metal (packbench 6/6), Apple OpenCL (6/6, same fingerprints) and the CUDA emulation (6/6); re-exporting with the width pinned at 4 B and rebasing onto e752fc7; PC job in about 20 minutes. Two class bugs found and fixed on its branch, both outside its layer: (1) proto-cuda/nvrtc/packfile.h re-derived the seed words as attempt 0 only, so ANY pack with IGNEUM_PROGRAM_ATTEMPT >= 1 (5.14% of chain epochs under v2) is refused by the one-click workers with "the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT", the error PC 1 logged on 5 October and attributed to the export race; fixed attempt-aware, with a tampered-attempt refusal test. This fix must ship in the 0.3.11 workers whatever else does. (2) proto-cuda/host.cu and proto-opencl/host.c derived dataset words on the host as mh_item(w >> 4)[w AND 15] instead of the pack's mh_word; fixed.
Consequences review (a20f8c09b90016cc7) C1, C10, C11: C1 recorded in the rollout plan 7a (0.3.11 on every prover before 16:00Z on 6 October, else the fee switch is republished at tip + 86,400); C10 was resolved by the mixer x4 decision (20:23, 20:24 entries), the pool-core and node verification numbers are being measured on ca2-mixer; C11: AMD RDNA 4 is about a seventh of a 5090 on this hash by dependent-read rate, 4.9x the electricity per hash and 2.2x worse per pound at list prices (0.032 against 0.072 MH/s per pound, read-width.md 4.1, approximate); the card's memory system, not a tuning gap; MH/W and MH per pound columns (list prices, approximate) go into the final table and the level 3 page; the line-width question is a 3.0 question since no width closes the gap without making the 5090 bandwidth-bound.
## 20:29 C1 decided for the morning; one packfile.h fix for 0.3.11
C1 (the fee switch H = 210,000 at about 18:45Z on 6 October by the 22:32:54Z DAA read, 1.002 DAA/s since 15:40Z; a 0.3.10 prover's shards are vetoed from H because its export RPC carries no daaScore and no fee schedule). Decision: (a) 0.3.11 carries the proving v1 fork eb32c645 (on 21d4c73c; protocol 15, message 75) and the app branch b399708 (on 5b0d54f) with H unchanged at 210,000, published as the fleet sweep. THE 16:00Z CHECK on 6 October, for whoever holds the morning: open the console's machine cards; if every prover (PC 1 ae432dc7, PC 2 1ccfe586, the Mac) shows 0.3.11, H stands and nothing is done; if any prover is not on 0.3.11, the publisher republishes H = tip + 86,400 by the fee-switch plan's rule (one digest flip, every node in one sweep: manifest, update-now, hand nodes, seed) before about 18:45Z. The proving agent recommends (b) unless (a) is certain; the check decides it.
The attempt-0 packfile.h bug (the era agent's find) is the one that took the fleet down at 18:23Z (epoch 34, attempt 1); it is already fixed attempt-aware on pack-loop af983a7 and merged into the 0.3.10 tree. Rule for 0.3.11: one derivation, one test set: the era and cache agents build their workers on that packfile.h and keep only tests that add a case; the host.cu / host.c mh_word derivation fix (new, the era agent's) stays with its own test.
## 20:31 card lifetime merged; cache freed after the build; growth mapping (b) recommended
`docs/analysis/card-lifetime-2026-10-05.md` (branch card-lifetime 1fecfe2) merged into ca2-coord. Decided (delegated): the GPU frees the 256 / 512 / 1,024 MiB cache after the daily dataset build (the hash never reads it; the rebuild costs 0.67 ms fill + 13.4 ms build on the 5090 at x1, about 54 ms at x4, owed); hot-table.md's resident reading is corrected, era-layout.md's freed reading stands. Recommended for Josh: growth mapping (b), power-of-two steps at years 4, 12, 28, 60 with AND MASK, the only mapping under which the litepaper's "4 GB about four years, 8 GB more than a decade" holds (4 GB: year 4 under (b), 1.0 to 1.5 years under (a); 8 GB: year 12 or 6.3 to 7.5; 12 GB: year 28 with the cache freed; 24 GB: year 60). Public lines and the evidence row go on the integration branch (rollout plan 6c); hot-table.md line 73 (the 8 GB row counted a 5090's warps) goes to the cache agent.
## 20:32 layer 9 agent started; the dot4 probe is on PC 1
ca2-epoch (a32a3ece66c02417a): the epoch length as an era parameter (600 to 7,200 DAA s, base 3,600; draw or 90% signal; the VDF rule; the difficulty-window constraint; the FPGA threat with citations), Mac compile-ahead measured now, the 5090 and 9070 XT compile times cited from the bench log, docs/plans/epoch-length.md. The dot4 probe job is running on PC 1 (ca2-analysis tip ee42d7c carries the playbook; the 5090 confirmation is the agent's watch). PC 1 queue after it: the era six-pack job, then the hot-table added-form job. PC 2: the proving agent's, then the ca2 node suites.
Agents running: ca2-era, ca2-cache, ca2-node, ca2-mixer, ca2-epoch; ca2-analysis watching its PC job. Done: ca2-soundness.
## 20:38 layer 7 complete on all three cards (ca2-analysis ee42d7c); PC 1 free
dot4 probe on PC 1 (jobs fetch-dot4-20261005, run-dot4-20261005, exit 0, 101 s; both cards restored and mining; 5090 SM clock 2,505 MHz before and after):
| Device | ALU chain, G steps/s | signed dot4 emulation, G dot4/s | dot4 instruction, G dot4/s | emulation vs instruction |
|---|---|---|---|---|
| RTX 5090 (NVIDIA OpenCL 3.0, driver 617.14) | 8,753.5 | 1,239.1 (7.1x an ALU step) | 7,453.6 (inline PTX dp4a.s32.s32, 1.17x) | 6.0x |
| RX 9070 XT gfx1201 (AMD-APP 3683.0) | 701.4 | 480.8 (1.46x) | 664.3 (__builtin_amdgcn_sudot4, 1.06x) | 1.38x |
| gfx1036 (RDNA 2 iGPU) | 40.6 | 15.8 (2.6x) | sudot4 does not build (needs dot8-insts) | |
| M5 Max (Metal) | 879.8 | 188.2 (4.7x); unsigned 548.2 (1.6x) | none exists | |
All bit-exact against the CPU reference. One dp4a costs about one ALU step on NVIDIA and AMD; the 5090 is 12.5x the 9070 XT on the ALU chain and 11.2x on dot4 (the family does not widen the AMD gap), 10x the M5 Max on the chain and 13.6x on dot4 (Apple's emulation widens its gap 1.4x). At W_new 4 the family adds about 21 ops per hash per lane; hash-rate losses expected under 5% on every card (to be measured with the family live). Owed: the CUDA __dp4a cross-check (needs nvcc), Metal 4 matmul2d int8 on the M5, sdot4 on RDNA 2. cl_khr_integer_dot_product is listed by no driver we own.
PC 1 is free: the era six-pack job goes next when its package arrives, then the hot-table added-form job.
## 20:39 PC 1 scheduler (the coordinator's role from now): the queue
Rule: one PC 1 job at a time; an agent asks by message before publishing, gets a "go PC 1" from this coordinator, and reports when its RESULT lines are in and both cards are restored; a hash-rate or power number taken while another job holds a card is not a number. The CPU-only job runs only in a slot where no measurement overlaps it.
| # | Job | Agent | Cards | Length | State |
|---|---|---|---|---|---|
| 1 | Era six-pack (layers 4 and 8), 5090 and 9070 XT | ca2-era a452664c512c73b9b | one card at a time | about 15 to 20 min | waiting for the package (packs on the pack-loop packfile.h) |
| 2 | Hot table, added form, probe 32/64/96 MiB plus packs | ca2-cache a5271cf269757b118 | one card at a time | about 15 min | waiting for the re-measured Mac rows and the rebuilt zip |
| 3 | Reproducible benchmark run | a0b9f574775ef1693 | one card at a time, 120 s per card | about 5 min | queued |
| 4 | AMD sweep on the 9070 XT (core clock and power steps) | a01dcb34ae16d867c | 9070 XT only; the 5090 keeps mining | about 20 min | queued |
| 5 | Ember Tune end to end, both cards | a855dcc4bd05e0615 | both | to be stated | queued after the AMD sweep |
| 6 | AMD-proving CPU fallback (CPU-only SP1 run, both cards mining) | a39db54d4de4af51e | none; loads the CPU | to be stated | last, or in a gap where no measurement runs for its whole length |
If job 1's package is more than 15 minutes away when job 2 is ready, job 2 goes first; the short job 3 fills any gap of under 10 minutes between packages.
20:39. No more agents are spawned tonight (Josh: no unnecessary credits); the running ones finish. PC 1 queue change: the AMD-proving CPU fallback's small fixture (block-56-transfers-3shards, minutes) runs NOW in the gap before the era package; its S_p shard (block-338-shard1, up to 30 min of every core) stays job 6, last. Next-cut list gains the rig installer's two follow-ups (the Linux manifest entry, the Linux prover build), tied to whichever release carries the proving half (rollout plan 8a).
## 20:42 the node switch is written; the Metal worker is a gate item
ca2-node (a3f505a9d981300cd): ca2-v3 commits d2cd6e1 (pack-loop af983a7 merged: packcheck.rs and the attempt rule; one packfile.h conflict resolved), 50d5c86 (the seam: ProgramClass { V2, V3 }, V3_CLASS placeholder w16, generator 3 in the program id, Epoch::from_chain_seeds, Epoch::chain_program, IGNEUM_PROGRAM_CLASS and IGNEUM_ERA_SEED_HEX in the packs, packcheck refuses wrong class / era / generator; v2 packs byte-identical; 53 tests), 9ed787e (workers: packfile.h reads class and era, CUDA and OpenCL workers take `class=v3 era=<hex>` tokens on job and prepare lines, pair identity includes them, mismatch answers `need`; 13 packfile checks). Fast-time gate script infra/fast-time/class-v3.mjs written, not run (needs the fork binaries). Node (ca2-v3-node, uncommitted until cargo check passes, queued behind the measure lock): the field in Params, OverrideParams, override_params and the digest (unconditional, its own statement; 11-entry digest test), the daemon line, program_class_for_epoch_at (v3 iff 3600 e >= N4, first epoch = ceil), POW_ERA_BLOCKS 15,552,000 and POW_ERA_LEAD 7,200 with the era stand-in (era 0 = genesis), EpochSeeds { epoch, day, class, era }, template and RPC fields 12 to 16, the miner's job line and seeds.txt. PC 2 command ready (run from the ca2-v3 worktree so push-build-inputs.sh packs the v3 igneum-pow); held until "ready for PC 2".
Gate item found: main.swift refuses v3 lines "until Swift has generator 3"; the Mac worker never regenerates a program (igneum-miner export-pack writes the pack), so the fix ordered is to accept the pack's class and era against the line, as the CUDA and OpenCL workers do. Without it Mac node 1 and the Mac app cannot mine v3 (gates G1 and G4).
PC 1: the AMD-proving small fixture is running; Ember Tune's 8-minute CPU-only build takes the next CPU gap, its 30-minute both-cards run is job 5.
20:43. Consequences round 2 (C18 to C20). C18: "near parity per pound" was wrong and is struck everywhere; read-width.md section 4.1 gives the 5090 at 2.2x the 9070 XT per pound at list (0.072 against 0.032 MH/s per pound, approximate), 4.9x per watt, 7.5x in rate; the level 3 page carries those. C19 (to ca2-mixer, already sent by the reviewer): the x4 verifier cost in IBD minutes over the 108,000-header pruning window per tier (8.6 min against 1.1 on one M5 Max core at the top of the range), pool shares per core per second, a scaled 2019-class figure (approximate), and the 10 ms gate margin left for 3.0 go into mixer-x4.md; the seeds' header-verify load goes into the testnet go checklist. C20 (to ca2-epoch, already sent): both rig miners run --exit-on-seed-change and re-export on exit 42, so a 10-minute epoch restarts every card's miner six times an hour and the Mac fleet's prepare pause goes from 35 s to 3.5 min an hour; the epoch-length document gets a per-tier restart-cost row and the compile-ahead margin against the VDF at 600 DAA s, and the rig installer drops the exit-42 path for prepare-ahead before any short epoch can be drawn (next-cut list).
## 20:43 the Metal worker's v3 path; the app flag; the seam
Correction to the 20:42 entry: igneum-bench --serve DOES regenerate every program in Swift (serveProgram calls generateProgramV2), so a v3 line could not be trusted blind. ca2-node added servePackProgram in main.swift: a `prepare <e> <d> <dir> class=v3 era=<hex>` line compiles program_bound.metal from the pack the miner wrote (--prepare-packs) after the packfile.h checks in Swift (generator 2 or 3, class against generator, seed bytes against the line, IGNEUM_SEEDW_INIT against attempt_words, class and era against the line); the program store keys on (seed, class, era); a v3 job with no resident pack answers `need` + `error ... program class mismatch`; v2 lines unchanged. A pack program never races variants (the Mac loses the variant race on v3 epochs; its cost is the race's gain, from the miner-perf entry, to be quoted). Integration item: the Mac app and Mac node 1's miner command must pass --prepare-packs, else the first v3 epoch on the Mac worker ends in `need` lines; check app/igneum-app/src/engine.rs. The gate network's Mac miner is the CPU miner (igneum-miner --engine igneum-pow), unaffected.
The seam the node relies on (kept by every branch): Epoch::from_chain_seeds(epoch, day, era, class, label), Epoch::chain_program(epoch, era, class, label), Epoch::chain_dataset(day, class), generate_from_seed_bytes_program_class(label, seed, class, era), ProgramClass::{load_class, generator_version, from_generator, name, parse}, V3_CLASS, Program::era_bytes, packcheck::verify_pack_dir_chain.
Mac build queue: the measure lock has been held by a packbench run since 20:31Z with three build slots held and five builds waiting; the node's cargo check and the swiftc recompile wait behind it. This is the lock working as designed; it sets the pace of the gates tonight.
## 20:45 the 9070 XT has dropped off PC 1's bus; app restart facts; the era package ETA
The AMD sweep agent (a01dcb34ae16d867c) reports from a read-only probe at about 20:40 UTC: Get-PnpDevice lists only the integrated "AMD Radeon(TM) Graphics" (gfx1036) and the RTX 5090; the app's AMD worker now mines the gfx1036 at 3.12 MH/s; the 9070 XT is absent from PnP (the eGPU link: the Sonnet box or the USB4 router; earlier today it went Code 43 and came back after a driver reinstall and reboot). A 10-second rescan probe is granted (pnputil /scan-devices, the USB4 router status). Josh is asleep and is not woken. Consequence if the card stays absent: gate G1 (bit-exact v3 on all three cards) and the 9070 XT rows of the era and hot-table tables cannot be taken tonight; the AMD-vendor stand-in available is the gfx1036 (RDNA 2, AMD OpenCL 3683.0, 3 MH/s), which ran the version 1 and version 2 conformance; whether it satisfies G1 for the devnet publish is asked of the coordinator. Every 9070 XT row taken before 20:40 (readwidth, dot4) stands.
App restart facts from the log intake (`node tools/logs.mjs`, 20:45): PC 1's app run is win-ae432dc7-20261005-190232 (started 19:02:32, no restart since), so no PC 1 measurement tonight straddled an app restart; PC 2's run is win-1ccfe586-20261005-200114 (started 20:01:14, before the readwidth 5090 round at 20:09). The 0.3.10 manifest is still unpublished (the shipper's CI is queued); the "jobs folder cleared by the 0.3.10 update" reading was wrong: the folder is cleared by fetch jobs.
The --prepare-packs item of 20:43 is resolved: app/igneum-app/src/engine.rs line 1225 passes it on master and on the 0.3.10 tree.
Era package: 10 to 15 minutes away (the pack-loop packfile merged over readwidth's; the OpenCL verification and the mingw rebuild queued behind three held build slots); zip ~/Desktop/igneum-ca2-era-pc1.zip, fetch id fetch-ca2-era-20261005, one playbook relay/playbooks/ca2-era-pc1.ps1 doing both cards (about 6 to 10 min). Widths pinned at 4 B in every era pack (512 B per hash), windows identical across the six packs, so the six-era spread isolates stride plus interleave. The CPU-only proving fixture holds PC 1 until about 21:15 to 21:30; the hot-table job is not ready either, so the order stays era, then hot table.
## 20:46 ruling on G1; hardware event recorded
Ruling (coordinator): gfx1036 satisfies the AMD vendor for gate G1 tonight (a compiler-and-ISA property; it carried the v1 and v2 conformance); the 9070 XT's hash-rate and power rows are owed and taken when the link is back; every 9070 XT row before 20:40 UTC stands. Nobody is woken, PC 1's app is not restarted. The event is in the rollout plan section 7b (hardware events) for the morning summary: the second eGPU link fault today (Code 43 at install, a bus drop at about 20:40); Josh reseats the USB4 cable and the eGPU power; the 0.3.10 hot-plug code shows "removed" and picks the card up without a restart. The publish proceeds when every other gate is green.
## 20:47 the Sonnet box is off the link; PC 1 facts corrected; the queue after the era job
Rescan at 20:45:34Z (relay probe #203, 10 s): the 9070 XT stays absent after pnputil /scan-devices; the USB4 list shows only the host and root routers, the Sonnet Breakaway Box 850T5 router present at 17:18Z is gone: the box is off the link, not just the card. Job 4 (the 9070 XT sweep) is dropped, its rows owed with this reason and time. The era and hot-table PC jobs run their AMD half on the gfx1036 for bit-exactness only (the G1 ruling); their 9070 XT hash-rate and probe rows are owed.
Correction to the 20:45 entry: PC 1's app is 0.3.9 (file 15:47:20Z) and its process started at 20:01:14Z (pid 12340), a restart, not a 0.3.10 install; the log intake's run id dates the log file, not the process. Both PCs restarted at about 20:01Z, before every readwidth PC job (from 20:02:51Z) and the dot4 probe (20:27Z), so no measurement tonight straddled a restart. 0.3.10 is still unpublished.
PC 1 queue now: (1) the AMD-proving small fixture (running, release expected 21:15 to 21:30), (2) the era job (both halves, about 6 to 10 min), (3) the 5090 power-limit sweep (575 / 460 / 400 / 400 W, 90 s each, cap restored to 431 W, about 8 min; SM and memory clocks in the RESULT lines), (4) the hot-table job, (5) the reproducible benchmark (5 min), (6) Ember Tune's 8-minute build in a CPU gap then its 30-minute both-cards run, (7) the AMD-proving S_p shard (up to 90 min, CPU only).
## 20:49 mixer construction written; the measure lock cleared; PC 2 and the merged-tree suites
ca2-mixer (af345b1e2c541ffbb), no commit yet (lands when the crate tests and the v2 pack diff are green): LoadClass gains mixer_mult (1 or 4) and growth; LoadClass::MX4 = v2 loads, mixer x4, growth on, no width roll, so its program stream is version 2's draw for draw; memhard::Shape { mixer_mult, cache_log2_words } in MixParams; derive_items applies the mixer with keys round_key(r x m + j), j in 0..m, before each of the 8 reads and round_key(8m + j) after; Cache::fill_log2; growth_doublings(d) = ilog2(1 + d / 1460) (doublings at years 4, 12, 28, 60), cache_log2_words(d) = 26 + doublings, dataset_log2_words capped at 32; d = 1 on the devnet pack keeps 2^26 and 2^28. Emitters emit the m-loop only when m > 1 (v2 text byte for byte otherwise); program.h carries IGNEUM_MIXER_MULT and IGNEUM_CACHE_GROWTH. Seam addition (additive): Epoch::chain_dataset_day(day_bytes, class, days_since_genesis, genesis_dataset_log2); the node agent was told to wire the genesis day index from Params.genesis.timestamp and a pow_genesis_dataset_log2 field (28 on the devnet, in the digest). V3_CLASS becomes LoadClass::MX4 composed with the era and hot fields at integration. Numbers follow the lock.
The Mac measure lock: the packbench loop was the readwidth agent's (Metal currentAllocatedSize at 32 and 128 KiB scratch for the consequences reviewer's C12, not a decided row); it stopped the loop at about 20:52, so the queued builds (the node's cargo check, the mixer's tests, the epoch and mixer measurements) proceed.
PC 2: spcurve-stopped-pc2-pv1b closed 20:47:46Z (the card alone: an empty shard 13,875 MiB in 2.1 s; a v1 shard 20,435 MiB, 4.2 s; 2.25 M pgas 28,371 MiB, 6.6 s; 4.5 M 28,307 MiB, 8.5 s; the prototype 6.75 M 28,275 MiB, 11.2 s: the peak plateaus at 28.3 GB from 20 M cycles up); spcurve-miner-pc2-pv1 (the same with the miner on, about 5 min) runs now; PC 2 is released after it. The proving fork tip is 3203c8d0 (eb32c645 plus the pool test's field and N = 8), app 440fd59; its unit tests ran on the Mac (consensus-core 13, exec 8, flows, 0 failed); its harness runs on the Mac. Decision: ONE PC 2 suite job on the merged tree (ca2-v3-node plus 3203c8d0) once the ca2 node branch is committed, covering both halves of 0.3.11.
## 20:50 the measure lock holder is a prover measurement; a lock-status defect
Correction to the 20:50 entry above: the measure lock has been held since 20:31Z by pid 45000, a `with-lock.sh measure` of igneum-wt-agg-cost's igneum-prove-host (the aggregation-cost agent's prover measurement, 18 minutes so far), not by the readwidth packbench; `with-lock.sh status` prints the LAST WRITER's command text, not the holder's, which is why it named the w4 run. Defect for the next cut (tools/lock/with-lock.sh: the status line must read the holder's pid and command, not the last writer's; the class of CLAUDE.md's watcher rule). Builds proceed in the three build slots (build-0 taken at 20:49:51 after an 885-s wait); GPU measurements (the mixer's verifier and build timings, the epoch compile-ahead, the Mac bit-exactness runs under `run` are not blocked) queue behind the prover measurement. Readwidth head is 30ff674 (per-watt rows for the consequences reviewer; e752fc7 stays the table commit).
## 20:51 the number-free public copy is applied on ca2-coord (0ad70ba and the next commit)
site/litepaper.html: the Mining section's "bound by memory bandwidth" is corrected to "waits on memory latency, not on maths or bandwidth"; the "Everything above is automatic" paragraph is replaced by the level 2 three ideas and the fourth paragraph with a link to the numbers page; the vs RandomX rows "Changes over time" and "Dataset" and the Hardware paragraph carry the step schedule (years 4, 12, 28; 4 GB about four years, 8 GB about twelve). site/index.html: the Memory row and the Mine card carry the step schedule ("2 GB at genesis, doubling at years 4, 12 and 28"; "any 4 GB card at launch, 8 GB from year 4"). NOT yet applied, because they carry the chip number: the level 1 sentence in the hero and the abstract ("a custom chip gains under 2x") and the limits bullet "A chip is impossible"; they wait for the combined chip row from docs/analysis/chip-model-v3.md (if 1.8x: "under 2x" with the margin stated as thin; else qualified). The bench page's Counter ASIC section (level 3) waits for the final table. docs/evidence.md's card-lifetime row (designed) is still to add.
## 20:51 the node builds every day cache through chain_dataset_day
ca2-v3 6c75dad (node agent): Epoch::chain_dataset_day(day_bytes, class, days_since_genesis, genesis_dataset_log2) with a placeholder body (the mixer branch fills growth_doublings under that signature), verify::days_since_genesis; the Metal worker takes v3 from a pack (compiled). Fork (uncommitted, in the cargo check holding build-1 since 20:50Z): Params::pow_genesis_dataset_log2 (28 on every network, in the digest in its own statement), Params::genesis_day_index(), install_pow_genesis in the daemon after the class switch, the engine's build_day through chain_dataset_day, the pack export through the same build_day, PowEpochInfo / RPC / proto fields 17 and 18 (genesis_day_index, genesis_dataset_log2; an old node's 0 reads as 28), override-60x.json and the redteam override carry pow_genesis_dataset_log2 28. Next: the check result, then "ready for PC 2" with the fork commit. evidence.md row 22 (card lifetime, designed) added on ca2-coord (3dc29fe).
## 20:53 spec text applied for the decided layers (ca2-coord 9b1f849)
docs/spec/01-lottery-hash.md: 1.12 carries epoch_len (the ladder 600 to 7,200, 90% signal at a day boundary, T_epoch and the lead fixed) with 3,600 unchanged on every network; 1.13.1 gains the mixer_mult row (4 under class v3) and the epoch_len row with the signal rule, the FPGA threat and the 600-s floor, and a pointer to era-layout.md for the layer 4 and 8 rows; 1.13.2 carries reserve family R1 = mm8 (uint8, W_new 4, unlock era 4 or 90% signal, the edge vectors, the native paths) and the emulation rule (8x per op, 5% hash-rate cap); 1.13.3 carries the cache growth rule (option C, growth_doublings(d) = floor(log2(1 + d / 1,460)), 256 / 512 / 1,024 MiB at genesis / year 4 / year 12, fill 0.2 / 0.4 / 0.8 s), the step mapping (b) as recommended, the shipped-density reason, and the cache freed after the daily build. docs/spec/04-seeds-and-vdf.md 4.3: the lead and T_epoch fixed at every epoch length. Still to land in the spec from the branches: 1.8.5 (the mixer x4 form, from mixer-x4.md), the 1.13.1 rows for stride, interleave and the window (era-layout.md), 1.5 and 1.8 for the hot table in the added form (hot-table.md), 1.17 and 1.15 for the v3 vectors and the conformance runs, 1.4.5 and 1.4.6 for generator 3 and the class in the pack.
## 20:56 PC 2 released; the proving fork tip for the merged tree
PC 2 is free (the proving agent's last job closed; the live prover is back on). Proving fork tip ece42979 on 21d4c73c (N = 8, the digest test edit), app 440fd59 or later; its suites ride with the ca2 node suites on the merged tree; its harness on the final tree is running on the Mac. The S_p curve with the miner on the card (PC 2's 5090): empty shard 15,585 MiB 7.5 s; the adopted v1 shard (30,000 pgas, 4.7 M cycles) 22,210 MiB 13.2 s (20,435 MiB, 4.2 s alone); 2.25 M pgas 30,049 MiB 17.9 s; 4.5 M 29,954 MiB 26.3 s; the prototype shard 30,083 MiB 33.3 s. Tiers as the proving agent published them: 32 GB mines and proves today, 24 GB from the fee switch (2.3 GB spare on the adopted shard), 16 GB empty shards only, 12 GB nothing on this build (D2 to Josh).
PC 2 queue: the ca2 node suites on the merged tree (ca2-v3-node + ece42979) as soon as the node agent sends "ready for PC 2"; nothing else is queued on PC 2.
## 20:57 the mixer construction is committed on ca2-v3's base
ca2-mixer 0fc0ad1 (rebased onto ca2-v3 6c75dad): LoadClass::MX4 = V3_CLASS (v2 loads, mixer x4, the growth rule; a class v3 program is the v2 program of its seed instruction for instruction, generator 3 in its id); chain_dataset_day has its real body (Shape::for_class_day: cache 2^cache_log2_words(d), dataset 2^dataset_log2_words(D_0, d)); 44 lib + 11 pack tests green, the two pinned v2 packs byte for byte. Next from it: --program-class v3 / --era-hex on the CLI, the pinned v3 packs mx4-genesis and mx4-devnet-epoch0 (era = the genesis-hash stand-in), the design and 1.8.5 spec text, the chip-model row; bit-exactness runs under the run lock now; the verifier and build timings wait for the measure lock (held by a live prover measurement from another worktree, pid 45000, 25 min in at 20:56). The node agent merges ca2-mixer before its fast-time gate, so the gate runs the real construction minus the era and hot fields.
20:58. PC 1: cpu-prove-pc1-small2 running since 20:54:02 (CPU only). PC 2: a fetch job from the aggregation-cost agent (job-fetch-prove-aggcost, 20:55:39) landed after the proving agent's release, so PC 2 is NOT idle for the ca2 suites until that agent's run closes; the suite publish checks `node tools/jobs.mjs status` for an idle PC 2 first. Readiness in hand: the repro benchmark (8 min, 5090 then gfx1036) waits for a gap; the 5090 power sweep (8 min) follows the era job; Ember Tune's build (8 min, CPU) and run (30 min, both cards) follow; the S_p CPU shard (up to 90 min) is last.
## 21:00 PC 1 released by the CPU fixture; the repro run has it; PC 2 to the aggregation-cost agent
cpu-prove-pc1-small2 finished 20:59:49Z: the SP1 CPU prover on PC 1 with the miners running: block-56-transfers-3shards shard 0 (200 pgas) 312 s wall, peak RSS 29.5 GB, 978% CPU; block-78-increment 322 s, 30.5 GB; the 5090 untouched (89% mean). The S_p shard job is DROPPED tonight: 312 s for 315 k cycles extrapolates the 60.8 M-cycle shard to many hours of every core and over 30 GB (approximate), which answers the CPU-fallback question (not viable for S_p shards; viable for empty or tiny shards only). "go PC 1" given to the reproducible benchmark (the 5090 then the gfx1036, about 8 min); the era job follows it, then the 5090 power sweep, then the hot table, then Ember Tune's build and run. "go PC 2" given to the aggregation-cost agent (20 min, GPU proving with the miner on then paused, prover restored); then the ca2 node suites, then the repro run's PC 2 slot (10 min).
21:01. GitHub Actions is in a major outage (six queued runs since 19:26Z, none acquired); the coordinator gave the 0.3.10 shipper the fallback at 21:00Z: build the Windows installer on PC 1 (MSVC window host, the payload under Git Bash, Inno Setup; CPU only, about 15 min). PC 1 order now: the repro run (until about 21:09), then the 0.3.10 installer build (the fleet's release, ahead of every measurement), then the era job, the 5090 power sweep, the hot table, Ember Tune. Any measurement that straddles the build window is re-run.
21:02. The Mac measure lock, found by `lsof`: the recorded holder pid 43916 is dead; the files are held open by two WAITERS, the readwidth agent's re-queued footprint loop (pid 78893, holding the measure and build files 11 min, waiting for the three build slots, which cargo tests keep re-acquiring: a convoy) and the epoch agent's compile-ahead measurement (pid 78476, waiting behind it). The readwidth agent is asked to kill 78893; the epoch measurement then runs when the build slots drain. Two defects for the next cut (one task filed): the status line shows the last writer, not the holder; a measure waiter can hold the master lock while build slots keep being granted to new builds, so a measurement can wait indefinitely under a steady stream of cargo tests.
## 21:04 proving v1 handoff received; the AMD-proving line on the site; amd-prove merged
Proving v1 for 0.3.11 (rollout plan 8a): fork ece42979 on 21d4c73c, harness PASSED (21 checks) in 244.4 s at 20:56:45Z on the final fork tree, override fields and the mixed-fleet rule recorded; the app's final hash follows its gate tests (90d3299 before it). Branch amd-prove (f1d7a7d) merged into ca2-coord (the append-only bench-log conflict kept both entries); its finding: no zkVM proves on an AMD GPU as of 5 October 2026 (SP1 CPU and CUDA; RISC Zero and ICICLE add Metal; nothing for AMD), the CPU fallback is about 5 minutes per small shard at 30 GB RSS, not a tier. The public line is applied on ca2-coord (1c8439f) with the proving agent's measured tiers in place of the doc's 16 and 20 GB: "Proving needs an NVIDIA card with 24 GB or more (32 GB until the fee switch of 6 October 2026; from it a 24 GB card mines and proves on the same card: 22.2 GB peak with the miner on). AMD and Apple cards mine. A prover for them lands when a zkVM ships one." It replaces "the card mines and proves" on the index, the litepaper's vs RandomX row and proving section, and the miner page (title, meta, hero, feature). Left as it was: the app's Proving tile text (the proving agent's).
## 21:05 the node merge for 0.3.11 is done; the hot-table package is ready; unproven_daa 10 at fast time
ca2-v3-node: 2e464e81 (the class switch, the era stand-in, the template, the miner) and the merge of proving-v1 ece42979 = ba43cf0f. One conflict, params.rs's digest-test edits array, resolved by keeping both sides (14 entries); every other shared hunk auto-merged as separate blocks. The merged tree is in cargo check (with igneum-exec and kaspa-p2p-flows); "ready for PC 2" follows with ba43cf0f once it and the Mac pow and consensus-core tests are green. Main repo ca2-v3: ca2-mixer fast-forwarded (66eeba3), then 43ca289 (the fast-time and redteam overrides carry the proving v1 fields and pow_genesis_dataset_log2 28; class-v3.mjs prints the dataset build ms per epoch). proving_v1_unproven_daa is a DAA clock (exec/src/proving.rs segment_status), so 10 at fast time is right (the proving agent confirms; its harness passes --unproven itself). The PC 2 suite command is the node agent's final shape (75-minute budget, six node crates plus the app tests), published from the ca2-v3 worktree when PC 2 is idle (the aggregation-cost job closes at about 21:21).
ca2-cache 196db96 (rebased on ca2-v3 464d6e1, the added form): 47 + 13 tests; the three added packs bit-exact on Metal and Apple OpenCL (fingerprints at 2^20, base 0: hot32k4a afb700b2d997c847, hot64k4a ba214baa9c1a9e85, hot96k4a 29e1916aed6deff5; 96/96; hot table PASS); first-pass rates under load (not numbers): 25.7 / 23.9 / 22.9 MH/s against v2 27.7 (the probe predicts 0.96 / 0.94 / 0.94); the measured rows wait for the measure lock behind the epoch measurement. PC package: ~/Desktop/igneum-ca2-hot.zip sha256 bd49faa1c9d48024f49c615481faff5c68a4c09f0889dbaf009c208674d67b3f (workers 956c4ab3... and 32d3d343... from 196db96, eight packs), fetch-ca2-hot-20261005, playbooks ca2-hot-5090-bench.ps1 and ca2-hot-9070-bench.ps1 (gfx1036 fallback). Its go follows the 0.3.10 build on PC 1 unless the era package is there first.
## 21:11 the 9070 XT is back; PC 1 to the 0.3.10 build; the repro run failed at parse time
The repro run (run-repro-pc1-20261005, 21:04 to 21:09:24Z) switched every card off and on and restored them, and found the 9070 XT (gfx1201) ON the bus again (opencl:1; the app switched it off and on), so the era and hot-table jobs run their gfx1201 halves as planned and the G1 ruling's fallback is not needed unless the link drops again; recorded in the rollout plan's hardware events. The run produced no numbers: repro.ps1 failed with a PowerShell parse error on each card (MissingEndParenthesisInExpression), the second playbook tonight that passed no local parse (the Mac has no pwsh); the class fix ordered: every PowerShell playbook parses itself on the PC as its first step (System.Management.Automation.Language.Parser::ParseFile, errors printed, non-zero exit), the way the dot4 playbook gates its bash body with bash -n. "PC 1 is yours" given to the 0.3.10 shipper at 21:10 for the installer build (CPU only, about 15 min); then the hot-table and era jobs, the 5090 power sweep, the repro re-run (about 22:00), Ember Tune.
21:20. Proving v1 app branch final: 6dc686a on 5b0d54f (provedefault 6 of 6; the app's 0.3.11 inputs are complete on that side); fork stays ece42979 (merged into ca2-v3-node ba43cf0f). The 0.3.11 app tree = the 0.3.10 release tree 5b0d54f + proving-v1 6dc686a + whatever the app needs for v3 (the --prepare-packs flag is already there; the Metal worker change is in the worker, not the app); the 0.3.11 main tree = master + ca2-v3 (igneum-pow, workers, fast-time, docs) + ca2-coord (the plans, the spec, the site copy).
## 21:20 layer 9 complete (ca2-epoch 4300608, e95e8b5)
Design (reserve-only): epoch_len base 3,600, ladder 600 to 7,200, SIGNAL ONLY (the era stream consumes draw 8 and ignores it); day-anchored epochs so a change lands in days; VDF option A (T_epoch 600 s and the 1,200-s lead stay genesis constants: the program is known 600 s ahead at every length, grinding margin 300x; option B rejected at 50x and a 200-s-deep checkpoint); REF_WINDOW_V2 = min(600, L); the 144-s settle per step is 24% of a 600-s epoch. Floor 600 DAA s: the slowest compile-ahead is the variant race at 38 s on the Mac and the 5090, 6.3% of the epoch and inside the window.
| Card | Compile-ahead | Share of a 600-s epoch |
|---|---|---|
| M5 Max Metal, race off (measured 21:18 UTC: 15.9 / 17.7 / 20.4 ms over 10 fresh programs, pack 79 ms then 1 ms) | 0.5 s (1.8 s cold) | 0.1% |
| M5 Max, race on (M11) | 38 s | 6.3% |
| RTX 5090 NVRTC, race off (M11, hot-swap) | 1.0 s | 0.2% |
| RTX 5090, race on | 38 s | 6.3% |
| RX 9070 XT OpenCL | 0.31 s + compile owed | |
| Intel UHD OpenCL (M11) | 6.4 s | 1.1% |
| Radeon iGPU under load (M11) | 124 s with the dataset | 20.7% (needs per-day dataset reuse before any signal below the base) |
FPGA citations: PRflow (FPT 2019) 42 min typical, 160 min worst for a monolithic Vivado compile; Aldec hours on Virtex UltraScale; partial reconfiguration milliseconds per region (ICAP 400 MB/s) shortens the load, not the compile; at 600 s a per-program bitstream mines 0% of each epoch, 47% at 3,600. Riders: the race defaults off (M11: base wins on both cards; 6.3% at the floor); per-day dataset reuse in the workers for the iGPU tier. Owed: the 9070 XT clBuildProgram time, the difficulty settle at a 15% step and 600-s epochs in sim.py, the 3-bit signal encoding against Kaspa's version bits (spec 5.8).
## 21:20 the hot table in the added form, measured on the Mac: the big-die chip comes out ahead
ca2-cache (hot-table.md 6.2 to 6.4): M5 Max, 21:03 to 21:19 UTC, load average 7 to 14, Metal packbench 2^24 x 5 (GPU time) and Apple OpenCL --bench-pack; v2 in the same session 27.63 / 27.59 MH/s.
| Pack | Metal / OpenCL MH/s | g against v2 (probe predicted) | Fingerprint (2^20, base 0) | Verifier ms per warp (v2 0.602) | Fill, one core |
|---|---|---|---|---|---|
| hot32k4a | 25.76 / 25.72 | 0.93 (0.96) | 8a3414735db4523c | 0.631 | 21.7 ms |
| hot64k4a | 23.92 / 23.87 | 0.87 (0.94) | 45668f34105f6307 | 0.609 | 43.3 ms |
| hot96k4a | 22.92 / 22.88 | 0.83 (0.93) | af763997dfee4c82 | 0.614 | 64.9 ms |
Reading: on Apple the added form costs 7 to 17% of the rate for four extra loads per iteration, more than the probe predicts as the table grows; only the 32 MiB table is near free. Chip arithmetic with these g: a chip serving H from DRAM keeps 0.86 / 0.92 / 0.96 of its gain; a 100 mm^2 die with the SRAM keeps 0.96 / 0.92 / 0.88; a 750 mm^2 die with the SRAM comes out 6 to 15% AHEAD, because the GPU pays the hits in rate and a big die pays them in 1.6 to 4.8% of area. So on the Mac's numbers layer 5 does not pass its own test; the decision waits for the 5090 (96 MiB L2) and 9070 XT (64 MB Infinity Cache) rows, where the hits may be near free (g close to 1). Rule for the decision: layer 5 goes into v3 only if, on every card we own, g is at or above 0.97 at the chosen size AND the on-die-cache chip row (chip-model-v3.md) moves down with it; otherwise layer 5 is out of v3 and stays a measured option for 3.0.
21:21. PC 2: agg-cost-pc2-1 closes at about 21:25Z (its own-miner phases ran the iGPU miner by a script fault; the curve is unmeasured); "go PC 2" given for agg-cost-pc2-2 (about 17 min, to 21:44Z), release due by 21:45Z; the 0.3.11 suites on ba43cf0f take PC 2 next (the node agent's Mac suites, release build and fast-time gate are in flight). PC 1: the 0.3.10 installer build (from 21:10); the hot-table job (package in hand) goes the moment the shipper reports the build closed; the era package is still being built.
21:22. Ember Tune (f9bf552 on ember-tune) is queued: its Windows app build (8 min, CPU) in the first gap after the installer build, its baseline run on the 5090 (8 min, no prompt; the 9070 XT if it can be taken without one) after the repro re-run. PC 1 order: the 0.3.10 installer build (running), the hot-table job, the era job, the 5090 power sweep, Ember's build in the first CPU gap, the repro re-run (about 22:00), Ember's run.
21:22. PC 1: the 0.3.10 installer build job (job-fb-installer-pc1) FAILED at 21:20:05Z, exit 2 after 2 s (the script, not the build); the shipper is asked to republish within 5 minutes or yield PC 1 to the hot-table measurement (10 min) and follow it. PC 2: agg-cost-pc2-1 still running at 21:21.
21:23. The shipper republished the installer build as fb-installer-pc1-2 (the first exit 2 was Test-Path on a \\wsl$ root path refused to the non-elevated session; the engine is now copied out with wsl -u root), cap 25 min, so PC 1 is the build's until about 21:50; the AMD presence probe (10 s, nothing held) runs beside it. Then: the hot-table job, the era job, the 5090 power sweep, Ember's build and fetches, the repro re-run, the AMD sweep (26 min), Ember's run (25 min).
## 21:23 the era package is ready; the V3_CLASS composition rule
ca2-era PC 1 package: ~/Desktop/igneum-ca2-era-pc1.zip sha256 f26f997602d94b0a408a6974ee484a7b3249ce18dc91dc00783aa9754e5ff040 (workers 5dd3bc16... and 9615efb3... from ca2-era on ca2-v3 464d6e1 with the pack-loop packfile.h merge and the host.c mh_word fix; packs v2 and era-0 to era-5, the six as class v3 chain packs, generator 3, program id 6b02c7c49eb126bd shared, width 4 B, the same devnet epoch seed and day); fetch-ca2-era-20261005; relay/playbooks/ca2-era-pc1.ps1 (the 5090 then gfx1201, gfx1036 fallback; 6 to 10 min). Mac bit-exactness on these packs: Metal 6/6, Apple OpenCL 6/6 (same fingerprints), CUDA CPU emulation 6/6. Its go follows the hot-table job, about 22:00.
Composition rule for the integration (three branches define V3_CLASS): V3_CLASS = LoadClass::MX4's fields (v2 loads, mixer_mult 4, growth on) + era: None (drawn per program inside generate_from_seed_bytes_program_class from the era bytes, LoadClass::era(V3_CLASS, era, &V3_ALLOWED)) + hot: None until the PC rows decide layer 5; the layout rides with the program (program.class.layout() in the interpreter and Epoch::dataset_word), so one day cache (chain_dataset_day) serves every era. The era agent rebases onto ca2-v3 HEAD with that literal; the node agent takes it into the integration tree. The era stream is seeded from "igneum-era/" || E_n (the index dropped: E_n commits to n through the VDF input); the spec text in era-layout.md says so.
21:24. The 9070 XT dropped off PC 1's bus again (probe #205 at 21:22:59Z: no Sonnet or USB4 router device; the third drop today; rollout plan 7b updated: the link is flapping). The era and hot-table jobs run their gfx1036 fallback unless the card is present at run time; the AMD sweep slot is conditional on a presence probe at 22:00. A probe reading of "miners off at 0.0 MH/s" was wrong: the app log shows the 5090 at 124.5 MH/s through 21:23:26Z; the AMD agent fixes its probe's precondition. The 0.3.10 installer build fb-installer-pc1-2 runs on PC 1 (cap 25 min from 21:23).
## 21:25 the mixer x4 construction is bit-exact on the Mac; the chip row reads 1.84x (thin)
ca2-mixer commits: 0fc0ad1 (construction, V3_CLASS = mx4, chain_dataset_day body), 66eeba3 (the pinned v3 packs mx4-genesis and mx4-devnet-epoch0, --program-class v3 / --era-hex), e4c04a7 (tests/mixer.rs: 200-program v3 fuzz, stats, edges, determinism; scratch.rs cherry-picked, 7 of 7), 7ce8d1e (docs/analysis/chip-model-v3.md). Spec text for 1.8.5 and 1.13.3 in docs/plans/mixer-x4.md section 2: the multiplied mixer with keys (r m + j + 1) x 0x9E3779B9; option C as doublings(d) = floor(log2(1 + d / 1460)); the day table with the verifier fill per step.
Bit-exactness (run lock): both v3 packs on Metal and Apple OpenCL, 3/3 standalone and 3/3 in batch, 96 of 96 lanes, dataset head / MASK / 64 samples PASS, one fingerprint per pack across both harnesses (6f48d5a2aa0dbe5f, 73caaebb28e808fe); the 200-pack v3 fuzz on Metal 200 of 200, every tenth on Apple OpenCL 20 of 20; CPU 44 lib, 12 packs, 7 scratch, 4 mixer tests; v2 exports IDENTICAL. Indicative (run lock, not a number): the Metal 1 GiB build at x4 30.2 ms (mx4-genesis) and 21.7 ms (mx4-devnet); the measured verifier and build timings wait on the measure lock.
The chip row (chip-model-v3.md), v3 at x4: 599,040 ops per hash; the on-die-cache recompute chip at 50 T op/s does 83.5 MH/s, 0.61x bare against the 5090's 136.1, 1.84x with the 3x fixed-function factor, 1.53x with the 128 mm^2 mirror deducted at equal silicon. With the hot table in the ADDED form at the Mac's g: 1.98x (32 MiB) and 2.12x (64 MiB) at equal budget, 1.60x / 1.67x with the SRAM deducted: the added hot table costs the card and not this chip, so it moves the row the WRONG way on the Mac's numbers. The claim holds "under 2x" on the equal-silicon convention, and on the equal-budget one only without the hot table; thin everywhere (a 3.3x factor or 10% on the budget reads 2.0x). Next lever: x8 (0.31x bare, 0.92x with the factor).
Consequence for layer 5: unless the 5090 and 9070 XT rows show g at or above 0.97 (the hits near free), the hot table stays OUT of v3 tonight (the rule in the 21:20 entry) and the public level 3 names it as a measured option, not a lever.
21:25. "ready for PC 2" from the node agent: fork ca2-v3-node 79bd8e10, Mac checks and tests green (rollout plan G6 row); the expected 0.3.11 digest with the two new fields at never is c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c. The coordinator publishes the suite job from the ca2-v3 worktree at 21:45 when the aggregation-cost agent releases PC 2, on the worktree's tip at that moment (the era and cache merges go in first if the era commit arrives in time).
21:27. The installer build's second attempt failed in 4 s on a path (the engine is not under /root/igneum-build/app/...); attempt 3 (fb-installer-pc1-3) is publishing with a find-based path; rule: if it fails inside 5 minutes, the hot-table measurement takes PC 1 before attempt 4. The shipper's jobs started no second app instance (the failed attempts exited before any exe ran); its post-build listing of igneum-app.exe processes settles the "off" reading.
## 21:27 decision rule for the mixer: x8 beside x4
Delegated (coordinator, Josh's "as strong as the measurements allow"): the mixer agent builds LoadClass::MX8 beside MX4, exports mx8-genesis and mx8-devnet-epoch0, runs the Mac bit-exactness, and measures v2, x4 and x8 in one measure-lock session (verifier ms per warp on one core, avg of 50 and worst cold; the 256 MiB fill; the Metal 1 GiB build); the 5090 and AMD daily-build times come from a 2-minute prepare job on PC 1 after the era job. x8 goes into v3 if the per-warp verify stays under 10 ms on one core AND the daily build stays under 1 s on every card we own; else x4 with the thin margin stated (1.84x with the factor) and x8 named as the next lever (0.92x with the factor, from the m16 table). The vectors are re-cut once after the choice. The hot-table rule stands (into v3 only at g >= 0.97 on both PC cards).
21:28. Installer attempt 3 failed at 21:27:20Z (exit 2 in 4 s: an inline bash -c string lost a quote through PowerShell; the fix is the 0.3.6 cut's: the WSL part as a file run with bash <file>, plus a read-only path probe before attempt 4). By the rule, "go PC 1" went to the hot-table job at 21:28 (10 min). PC 1 order from here: the hot-table job, the shipper's path probe and attempt 4 (about 16 min), the era job, the 5090 power sweep, the mixer daily-build job (2 min), Ember's build, the repro re-run, the AMD sweep (conditional), Ember's run.
21:31. Consequences round 3. C23: the x8 table gains a gfx1036 row (the integrated tier builds the dataset per PREPARE: 7 to 12 s at x1, 55 to 124 s under load, so 28 to 48 s an hour at x4 and 56 to 96 s at x8, every boundary missed) and a scaled 8 GB-class row; the "under 1 s" rule applies to the discrete cards' daily build; for the integrated tier either per-day dataset reuse in the three workers lands with v3 (the node agent is asked whether it is bounded tonight) or the level 3 page says the iGPU tier mines v3 with a restart per epoch; recorded with the x4/x8 choice. C24: two inline bash bodies lost a quote through PowerShell tonight (amd-prove's awk at 20:44, the installer's bash -c at 21:27); the reviewer's sub-agent adds a repo-wide CI check (every inline bash body through bash -n, an unextractable one fails CI) and the convention line in packaging/README-ship.md; no collision with the PC 1 queue.
## 21:31 GitHub's runner recovered: the 0.3.10 rollout starts; the restart rule for every PC job
The 0.3.10 Windows run 37374158235 went green at 21:30:29Z on the release tree; the three fb-installer jobs are done and expired; nothing more of the shipper's touches PC 1. The rollout runs now (manifest, update-now, hand nodes, seed, digest sweep): every app restarts once within the next quarter hour. Rules: the shipper is asked to hold PC 1's update-now until the hot-table job releases (about 21:40) and the era job starts only after PC 1's new STATUS line; the 0.3.11 suites publish on PC 2 only after PC 2's app shows 0.3.10 (a build job dies with the app); any measurement that straddles a restart is re-run; every PC job's RESULT lines carry the app version before and after.
## 21:32 per-day dataset reuse: Metal has it, CUDA and OpenCL do not (0.3.12); the gate re-runs after a script fix
Node agent: the Metal worker's ServeStore already keys datasets by day and programs by epoch (a prepare on a resident day builds the program only); the CUDA worker.cpp and OpenCL host.c bundle program, cache, dataset and the self-test in one Pair, and splitting a Day object out touches buffer ownership, releasePair, the prepare thread and the self-test in both: over an hour, not shipped untested tonight; first item after the publish (0.3.12; next-cut list). Decision recorded (C23): the level 3 page states that the integrated tier on the one-click workers mines v3 with a restart per epoch (public copy and rollout plan updated).
Gate G4: the first fast-time run failed at node start on the script, not the node: JSON.parse turned a never height (18446744073709551615) into 1.8446744073709552e+19 and the node refused the override; fixed as text merging in class-v3.mjs and simnet.mjs (d5ff532; no other script in tools/, infra/ or sim/ has the shape). The gate runs again on the mixer-x4 class (binaries from 79bd8e10 + ca2-v3 66eeba3). Main-repo tip for the suite job title: d5ff532 (era and cache not yet merged).
21:34. PC 2: agg-cost-pc2-1 closed 21:25:11Z (done, miners and prover back on); agg-cost-pc2-2 went out at 21:33Z (a missed close), self-limited to 16.5 min, closes about 21:51Z; the 0.3.11 suites publish after it AND after PC 2's app shows the 0.3.10 STATUS line (the update-now goes to PC 2 now). PC 1: the hot-table job runs (release about 21:40); PC 1's update-now follows the release; the era job starts after PC 1's 0.3.10 STATUS line.
## 21:36 layer 5 measured on both PC cards: OUT of v3
Hot-table jobs on PC 1 (fetch 21:29:01Z; run-ca2-hot-5090 126 s, closed 21:31:38Z; run-ca2-hot-9070 247 s, closed 21:35:39Z; both cards restored; the 9070 XT WAS on the bus, full path; app 0.3.9 throughout, no straddle). All eight packs bit-exact on the 5090 and the 9070 XT with the Mac's fingerprints.
| Pack | RTX 5090 MH/s (v2 136.1) | RX 9070 XT MH/s (v2 18.15) | M5 Max g |
|---|---|---|---|
| hot32k4 (replaced) | 146.6 (x1.08) | 19.79 (x1.09) | x1.22 |
| hot64k4 | 140.8 (x1.03) | 18.73 (x1.03) | x1.12 |
| hot96k4 | 138.5 (x1.02) | 18.33 (x1.01) | x1.05 |
| hot64k2 | 137.5 (x1.01) | 18.17 (x1.00) | x1.00 |
| hot64k8 | 163.6 (x1.20) | 22.32 (x1.23) | x1.71 |
| hot32k4a (added) | 118.7 (0.87) | 15.27 (0.84) | 0.93 |
| hot64k4a | 115.4 (0.85) | 14.62 (0.81) | 0.87 |
| hot96k4a | 114.4 (0.84) | 14.56 (0.80) | 0.83 |
Decision (the 0.97 rule): layer 5 is OUT of v3. Neither card keeps even the 32 MiB table resident while the 1 GiB dataset streams (the replaced form gains 1.02 to 1.08x at k = 4 against an ideal 1.33x), and the added form costs 13 to 20%; the chip row moves the wrong way with it. Layer 5 stays a measured option for 3.0 (a table small enough to stay resident, or a different access pattern). The probe rows and the writeup follow on ca2-cache. PC 1 is released to the shipper for PC 1's update-now; the era job starts after PC 1's 0.3.10 STATUS line.
21:38. ca2-cache final: 2de19e5 (nine commits from 55e285c, on ca2-v3 464d6e1); hot-table.md carries the probe rows for all three cards (5090 112.6 G loads/s at 32 / 64 / 96 MiB inside its L2 against 17.6 at 1 GiB; 9070 XT 9.88 / 9.47 / 8.18 / 2.43; M5 Max 21.7 / 12.8 / 12.3 / 3.50), the PC tables with g, the chip arithmetic at the measured g, the decision, the 3.0 note ("what would make it pay": a resident size found by a hash sweep below 32 MiB, k only with residency, a line-unit or streamed access shape) and the unverified list; the bench-log entry and two addenda carry the job ids and the worker sha256s. The probe promises full hits inside the 5090's L2 but the hash gets 2 to 8% at k = 4 because the streaming dataset evicts the table.
## 21:39 gate G4 run 1 PASS on the mixer-x4 class
Fast-time 3-node network, 21:33 to 21:38 UTC (fork 79bd8e10 + igneum-pow 66eeba3): the switch line on 3 of 3 nodes (active from epoch 3, DAA 150 rounded up to 180), templates class 2 then 3 from epoch 3, 181 blocks before and 124 after the boundary, program ids agree on all three miners (v2 e0 to e2, v3 e3 to e5), 0 rejected on miners and nodes, one sink on all three (082fd39ba65df2ff, 304/304/304), a new (day, class) cache 177 to 235 ms on one core. Main-repo tip bf04c56 (the doc, the summary JSON, the script's --connect fix). Run 2 on the composed class follows the era commit and the cache merge (rebuild about 10 min, gate 5 min). The x4 dataset build time comes from the mixer's PC prepare job (the CPU miner derives words from the cache).
## 21:40 x8 built and bit-exact beside x4; the mixer PC job retargeted to PC 1
ca2-mixer 504cae4 (LoadClass::MX8 "mx8", packs mx8-genesis and mx8-devnet-epoch0, the fuzz takes IGNEUM_MIXER_CLASS, playbooks with the x8 packs) and fe4e193 (mixer-x4.md per-tier build table and the x4/x8 rule; chip-model-v3.md with the mixer row as the headline, the layer 5 rows kept as measured not adopted with the PC g beside the Mac's; x8 rows 0.31x bare, 0.92x with the factor, 0.76x at equal silicon at year 0). x8 bit-exactness (run lock): both packs on Metal and Apple OpenCL 3/3 + 3/3, 96 of 96 lanes, one fingerprint per pack across both harnesses (7c28cfb06c5c65a9, bbb183f72692f840); 50-program x8 fuzz on Metal 50 of 50, every tenth on OpenCL 5 of 5. Indicative Mac builds (run lock): 30.0 ms at x8 against 30.2 at x4 (genesis pack), 22.0 against 21.7 (devnet pack): the Mac's build is latency-bound. The timing session (verifier v2 / x4 / x8, the fill, the build) is queued behind the measure lock. PC job: mixer-x4-pcjob.zip sha256 55a2913cb8790cd3b106dc3d0d29b6e2952b5a08378cd815935915ab82898a8e (v2 control plus the mx4 and mx8 genesis and devnet packs; a prepare per pack printing the worker's cache and dataset build ms); retargeted so both halves run on PC 1 (its 5090 and its AMD card), after the era job, about 22:05.
## 21:41 the mixer timing session: x8 passes the verifier half of the rule
M5 Max, one core, measure lock, 21:40:12 to 21:40:23 UTC, on a loaded box (load average 5.6 one-minute, 26 fifteen-minute: other agents' unlocked processes), so the absolute figures are about 2x the quiet 0.604 ms v2 baseline and the RATIOS are the measurement (two rounds, within 4%); a quiet-box re-run is owed for absolute numbers.
| Class | Verifier ms per warp, avg of 50 (round 1 / 2) | Worst cold unit | Ratio to v2 |
|---|---|---|---|
| v2 | 1.361 / 1.310 | 1.579 | 1 |
| x4 (genesis; devnet pack 1.923) | 1.956 / 1.923 | 2.043 | 1.45x |
| x8 (genesis; devnet pack 2.972) | 2.785 / 2.790 | 2.942 | 2.1x |
256 MiB cache fill on one core 172 to 175 ms. Metal 1 GiB build, GPU time: v2 21.0 ms (29.7 cold), x4 20.9 / 21.0, x8 21.9 / 21.9: the Mac's build is bound by the 8 dependent cache-line reads per item, not the arithmetic, so the "under 1 s on every discrete card" half of the rule is decided by the 5090 and 9070 XT rows of the mixer PC job (by the M16 arithmetic the 5090 is 54 ms at x4 and 107 ms at x8 if arithmetic-bound, 13.4 ms if latency-bound: far under 1 s either way). Verifier half: x8 passes with 7.1 ms of the 10 ms gate to spare on the loaded core (about 1.3 ms on a quiet core, approximate); x4 leaves 8.0 ms. Chip row at x8: 1,198,080 ops per hash, 41.7 MH/s, 0.31x bare, 0.92x with the 3x factor, 0.76x at equal silicon; x4 1.84x / 1.53x. C19 at these loaded figures: shares per core per second 735 / 511 / 358 (v2 / x4 / x8), a 22,000-member pool at one share per 10 s needs 3.0 / 4.3 / 6.1 cores; IBD over 108,000 headers on one core 2.4 / 3.5 / 5.0 min. Provisional choice under the rule: x8, confirmed when the PC build rows land (about 22:10).
## 21:41 the era draw passes the 5% rule on the Mac; the final era package; the merges for G4 run 2
ca2-era 9f98af2 (one commit on ca2-v3 HEAD; 48 lib + 17 integration tests; the pinned v2 packs byte-identical; mx4-genesis unchanged; mx4-devnet-epoch0 re-exported with the era inside the class). V3_CLASS = LoadClass { era: None, ..LoadClass::MX4 }; the era class is drawn inside generate_from_seed_bytes_program_class from the era bytes; chain_dataset_day and Epoch::dataset_word compose unchanged; an era program takes 11 draws per instruction. Mac (M5 Max, 21:38Z, Metal packbench 5 x 2^24, a loaded box): v2 27.68 MH/s; era-0 to era-5 28.58, 28.48, 28.35, 28.38, 28.49, 28.48: min 28.35, median 28.48, max 28.58, SPREAD 0.8% (under the 5% rule); 3/3 vectors and the in-batch vectors PASS on every pack; CPU verify 1.319 to 1.345 ms per warp against v2 1.334 in the same loaded run (quiet re-run owed). FINAL PC 1 package: ~/Desktop/igneum-ca2-era-pc1.zip sha256 f79c0607bb4187e3cf16fce3f533e7d525673d766d7edb799f27fd81af5dcee1 (workers 0fbfd50a... and 8c8caff7... from the merged tree; packs re-exported, attempt 0, program id 73bcbfe8ccf988f1 in all six with the era seed beside it); fetch-ca2-era-20261005; ca2-era-pc1.ps1. The 0.3.10 update-now reached every machine at 21:39:59Z (manifest live 21:33Z); the era job starts on PC 1's 0.3.10 STATUS line. The node agent merges 9f98af2 then ca2-cache 2de19e5 into ca2-v3 for G4 run 2 and the suites' tip.
21:42. ca2-v3 now carries the era draw: ca2-era's tip b105a55 (9f98af2 rebased onto the node agent's 88dafbc) fast-forwarded, no conflict; the suite job's main-repo tip is b105a55. ca2-cache 2de19e5 does NOT merge (it bases on 464d6e1, before the mixer and era commits rewrote the class literal, the load emitters, the pack fields and the pinned-pack tests: 8 files, 35 hunks); the node agent aborted cleanly and the cache agent is rebasing onto b105a55 as a squashed commit with hot: None kept; the composed class under test is unchanged by the cache code, so the igneum-pow and packfile checks, the igneumd and igneum-miner rebuild and gate run 2 proceed on b105a55 now.
21:45. PC 2 defect: since a job's /api/resume at 21:25:11Z the 0.3.9 app answered ok and never restarted the NVIDIA miner (nor the iGPU one): the 5090 worker "off" at hash 0 holding 1.7 GB, so agg-cost-pc2-2's mining phases are void (its idle phases run; closes about 21:52Z) and the devnet has been short PC 2's rate since 21:25. The 0.3.10 restart should bring the miners back; the shipper confirms PC 2's 5090 STATUS rate after the 0.3.10 line, else the aggregation-cost agent's restore script (tools/proving-v1/pc2-agg-cost-restore.ps1, 30 s) runs. Defect for the next cut: a resume that answers ok without a miner restart; the app must re-check the miner processes after a resume and report a failure. The aggregation-cost agent gets a 20-minute re-run slot on PC 2 after the 0.3.11 suites.
## 21:46 PC 1 is on 0.3.10; the era job has the go
PC 1 restarted on 0.3.10 at 21:40:41Z (engine run win-ae432dc7-20261005-214041), the 5090 at 141.4 MH/s by 21:45:17Z; "go PC 1" to the era job at 21:46 (fetch-ca2-era-20261005, zip f79c0607...; the 5090 then the AMD card; 6 to 10 min). The mixer daily-build job follows it, then the 5090 power sweep, Ember's build and fetches, the repro re-run, the AMD sweep (conditional), Ember's run. PC 2 is still on 0.3.9 at 21:45 (its restart pending); the 0.3.11 suites publish after its 0.3.10 line and a confirmed 5090 rate. Every measurement before the restart on PC 1 (hot table 21:29 to 21:35) stands: it did not straddle.
21:50. The resume fix (the miners not restarted after POST /api/resume: PC 2 since 21:25Z tonight, the Mac this afternoon) is assigned to the proving agent on 0.3.11's app branch (engine.rs resume: restart every enabled card's worker and a stale pack export, re-check within one tick, a state-machine unit test plus the known-failed case from PC 2's log); rollout plan 8a. If its commit is not in hand at the app cut, it heads 0.3.12's list.
## 21:52 gate G4 run 2 PASS on the composed class; the 0.3.11 suites are packing for PC 2
G4 run 2 (21:46:36 to 21:51:31 UTC, b105a55 = era + mixer, hot None): every check true; 181 / 124 blocks around DAA 180; v3 ids e3 2d278041ba482dba, e4 2ae786d294a8a59d, e5 bc36813df2f41b5f on all three miners; the v2 epoch-0 id 8f8806638d59850f unchanged from run 1; 0 rejected; one sink 712c1b212091dcdc at 303/303/303; the switch line on 3 of 3; cache ready 178 / 181 ms. G4 is GREEN. The 0.3.11 suite job is packing from the ca2-v3 worktree (fork 79bd8e10, main b105a55; build-inputs.zip 9,563,672 bytes sha256 bf89ab4c...) for PC 2, which is on 0.3.10 since 21:49:41Z with its 5090 worker back on the first try.
21:53. The 0.3.11 suite job is published: build-20261005-215219 to PC 2 (fork 79bd8e10, main b105a55; linux build 30 min, tests 35 min: kaspa-consensus-core, igneum-exec, kaspa-pow, kaspa-consensus, igneum-miner, kaspa-p2p-flows, igneum-app); PC 2 is on 0.3.10 with its 5090 back. The worktree freeze is lifted for the node agent (the run-2 summary commit, then the cache merge). The proving agent's resume fix waits on a test run behind the Mac's held lock slots.
21:55. The 0.3.10 rollout: both PCs mine on 0.3.10 with the rebuilt workers (PC 2 120.6 MH/s at 21:54:33Z, PC 1 141.3); the hand nodes (21:49:38Z, 21:49:50Z) and the seed (21:50:15Z) on 21d4c73c, digest 1f4b4425 everywhere; not yet on 0.3.10: the US laptop 37ba0461 (installer downloaded 21:40:52Z, app not back after 13 min; nothing to drive from here) and Sam's Mac (quit since 20:47Z). Open on PC 2: the prover fails with "CudaClientError: Connect(PermissionDenied)" since the restart (three shards 21:49:56 to 21:50:32Z); the likely cause is the sp1-gpu-server socket handling of the aggregation-cost jobs; the proving agent owns it and publishes a fix after the suite job (PC 2 is the suite job's until it closes). The ca2-v3 tip is 63dabb2 (run-2 summary and doc, the G6 job id recorded).
## 21:56 gate G6: the first PC 2 job failed on the known kaspa-consensus flake; split re-run
build-20261005-215219 (21:53:01 to 21:55:48Z): Linux build ok (igneumd 49,164,264 bytes sha256 11979b49..., igneum-miner d25a8270..., igneum-app 68007173...), igneum-app tests 78 + 26 + 8 passed; the node stage exit 101: kaspa-consensus 96 passed, 1 failed, processes::finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list, UnexpectedDifficulty(..., 487112384, 487129578) in mine_on_all. This is the flake the 0.3.10 cut met on the same base 21d4c73c under the six-package parallel run (release-0.3.10.md: it passed alone twice, job build-20261005-182804, 97 passed), a timing-dependent difficulty in the test helper, not a v3 change (v3 touches no finality code). Per the gate rule the publish stops here until the suites are green: job 2 of 3 (kaspa-consensus alone) is published now; job 3 of 3 (the other five crates) follows; the flake itself goes on the next-cut list (make mine_on_all deterministic under parallel load).
The proving v1 app branch is final at a223ca9 (6dc686a plus the resume fix with the PC 2 case as a unit test); rollout plan 8a updated. PC 2's prover fault is the root-socket class (agg-cost-pc2-1 ran the host as root in WSL2 and left /tmp/sp1-cuda-0.sock owned by root; the app's prover has failed every shard since 21:25:24Z); the proving agent's pc2-socket-fix.ps1 (60 s, miners untouched) runs between my two suite jobs.
## 21:57 the era PC job is done; a 2.2x CPU-verifier regression on ca2-v3 HEAD (gate item); the PC 2 order
Era job run-ca2-era-pc1-20261005 (exit 0, 301 s), both cards restored, the 9070 XT ON the bus at run time (gfx1201, 32 CUs): first rows 18.96 to 19.21 MH/s on the era packs on the 9070 XT, self-test PASS, the 2^24 fingerprints equal to the Mac's (era-4 3ace11ad84c053ae, era-5 a8897d82adceb4a1); the full 5090 and 9070 XT tables with the spread per card follow. "go PC 1" to the mixer daily-build job at 21:57.
GATE ITEM (found by the era agent, two binaries on the same v2 input, checksum 19297e99c7b9a55e, same minute, load 4 to 5): the CPU verifier on ca2-v3 HEAD (88dafbc) takes 1.332 ms per warp and on the era branch 1.310, against readwidth's binary at 0.604 and 0.606: the mixer branch's derive_items / Shape path costs 2.2x at m = 1, on the v2 path the live devnet verifies with. The mixer agent's "loaded box" reading of its 1.31 to 1.36 ms v2 figure was the code, not the load. Ordered: find and fix on ca2-mixer, restore v2 to within 5% of 0.604 ms measured the same way, re-measure v2 / x4 / x8 on the fixed binary (the C19 figures too), then the node agent merges it; the publish waits on it (G3's suite does not catch a slowdown; the 10 ms gate and the pool and IBD figures depend on it). Next-cut rule: a verifier benchmark with a pinned bound in the crate's CI.
PC 2 order: suite job 2 of 3 (kaspa-consensus alone, running), the proving agent's socket fix (60 s; the aggregation-cost restore is skipped), suite job 3 of 3 (the five other crates), the aggregation-cost re-run (20 min), the prover-floor agent's windows (through the proving agent). The proving app branch: a223ca9 is the last code change (8b47073 docs only after it).
21:58. ca2-cache is rebased: one squashed commit 1950661 on ca2-v3 63dabb2 (fast-forwardable; history under tag ca2-cache-history-2026-10-05); V3_CLASS = { era: None, hot: None, ..MX4 }; every hunk kept the ca2-v3 side and appended the hot code; 53 + 19 tests; the pinned v2, mx4, era and readwidth packs untouched; the eight hot packs re-exported with unchanged vectors, 96/96 and the pre-rebase fingerprints on Metal and Apple OpenCL. The node agent merges it after the mixer's verifier fix.
## 21:58 layers 4 and 8 decided IN on the PC rows (ca2-era 78c0ee4)
| Card | v2 MH/s | six eras min / median / max | Spread | Fingerprints = Mac | Latency-bound share |
|---|---|---|---|---|---|
| RTX 5090 (PC 1, CUDA, 1 warp per block) | 137.2 | 136.18 / 136.44 / 138.01 | 1.3% | 7/7 | 1.01 |
| RX 9070 XT (PC 1, OpenCL, group 256) | 18.09 | 18.61 / 18.93 / 19.21 | 3.2% | 7/7 | 0.95 |
| M5 Max (Metal, 5 x 2^24) | 27.68 | 28.35 / 28.48 / 28.58 | 0.8% | 7/7 | 1.06 |
All under the 5% rule; the CPU verifier 1.00 to 1.02 of v2 within one binary; the dataset build with the scatter store 20.7 to 21.8 ms against 21.1 linear on the M5 Max. Chip line: 512 B per hash, 120 to 128 distinct 64-B lines, the SRAM mirror is the whole dataset every hour (a windows-union census over 300 programs); the interleave buys nothing against a chip with a programmable address decoder (stated in the doc); the stride is a bijection with no cryptanalysis yet. Commits: b105a55 (implementation, history under tag ca2-era-pre-squash), c570da3, 669a27a, 78c0ee4 (the PC rows). The suite job 2 of 3 is build-20261005-215712 on PC 2 (kaspa-consensus alone).
21:58. ca2-v3 tip fbf958e: ca2-cache 1950661 fast-forwarded (no conflict), V3_CLASS = { era: None, hot: None, ..MX4 }; igneum-pow 53 + 19 tests, packfile-test 0 failures; the fork's check against the new crate running; the fork stays 79bd8e10. The node doc carries the CPU hash rate across the switch in run 2 (about 30% under v2 on the CPU interpreter, approximate, a shared Mac) and the verifier before/after slot for the mixer fix. Gate run 3 on the fixed tree follows the mixer fix.
## 21:59 0.3.10 is shipped and merged to master (cde561c, pushed 21:58:17Z)
The shipper's report: master cde561c (the 0.3.10 merge) + 1f0d62c (the plan); fork release-0.3.10 21d4c73c; digest 1f4b4425... on every node; six suites green on 21d4c73c on PC 2 with the same ban_is_decided flake under the parallel run (passes alone twice), recorded for the c4 agent. Open from it: PC 37ba0461 (the US laptop) stuck in its install since 21:41:16Z; Sam's Mac quit since 20:47Z; PC 2's prover dark (the root-socket cause is now named, the fix queued); C1 at 16:00Z; the /api/resume no-op (fixed on the 0.3.11 app branch).
Consequence for 0.3.11: the main tree base is now master cde561c, so the integration merge is ca2-v3 (8ea6740 plus the mixer fix) and ca2-coord into master, with the app branch a223ca9 (on 5b0d54f, which master contains). The fork base stays 21d4c73c (= release-0.3.10's tip), so the fork merge is clean by construction.
## 22:00 G6 job 2 of 3 green; the merge into master dry-run
build-20261005-215712 (21:57:12 to 21:59:45Z): kaspa-consensus alone 97 passed, 0 failed, 3 ignored (ban_is_decided ... ok), every stage ok. The PC 2 prover socket fix runs now (the proving agent, 60 s), then job 3 of 3 (consensus-core, igneum-exec, kaspa-pow, igneum-miner, kaspa-p2p-flows and the app tests).
Merge dry-run into master cde561c (a scratch worktree, aborted): ca2-v3 (8ea6740) conflicts in docs/bench-log.md and proto-opencl/host.c; ca2-coord conflicts in docs/bench-log.md, proto-cuda/nvrtc/packfile.h and proto-opencl/host.c (master's 0.3.10 merge brought pack-loop's packfile.h and opencl-rdna4's host.c). The bench log is append-only (keep both); packfile.h and host.c take the ca2-v3 side (it carries the pack-loop rule plus the class, era, mixer and hot fields) re-checked against master's hunks. The integration merge is the ship's first step.
## 22:01 the verifier regression bisected to the mixer's 0fc0ad1; the mixer PC 1 job is running; PC 2 queue
Bisection (mixer agent, measure lock, 21:59 UTC, same input 19297e99c7b9a55e, avg of 50, two rounds): readwidth e752fc7 0.607 / 0.609 ms per warp; the ca2-v3 seam 6c75dad 0.610 / 0.609; the mixer's 0fc0ad1 1.332 / 1.316; ca2-v3 HEAD 88dafbc 1.325 / 1.347. The 2.2x is in 0fc0ad1's derive_items / Cache path at m = 1 (not the era layout, not the load). Three candidate fixes building (the constant line mask back in Cache::line; an m == 1 fast path that is readwidth's loop verbatim; both); the one that restores 0.61 goes on top of ca2-v3 8ea6740 with the six mixer commits rebased, measured the same way; then the node agent merges, rebuilds and runs gate 3.
PC 1: fetch-mixer-x4-20261005 and run-mixer-x4-pc1-20261005 published 21:59:32Z (zip 152fcf93...; workers e6007918... and e4334aaf... from ca2-mixer; the five packs; 25-minute timeout). PC 2: the proving agent's socket fix (60 s) now, then suite job 3 of 3, then the prover-floor agent's toolchain check (3 min) and its 60-minute niced sp1-gpu-server build (CPU only; the shipped server panics on any card under 24 GB, sp1-gpu builder.rs 35 to 39, and allocates every prover at Setup: the 13.9 GB floor's cause), then the aggregation-cost re-run (20 min), then the prover-floor measurements.
0.3.11 ship template (release-0.3.10.md section 7): push the release branch, `gh workflow run windows.yml --ref <branch>`, `node tools/ship-app.mjs 0.3.11 --node <fork worktree> --branch <branch> --public --activation-height N4 --deadline-note "program class v3 + proving v1" --notes "..." [--from ci]` with the override object carrying every switch; gh must be on igneum-josh; the pre-push hook flips two site files (restore with `git checkout -- site/`).
## 22:03 PC 2's prover is back; suite job 3 of 3 published
socketfix-pc2-pv1 (22:01:14 to 22:02:12Z): /tmp/sp1-cuda-0.sock owned by root removed (the aggregation-cost job's run), the prover switched off and on; the next shard (block 89011 shard 0) "proven and submitted in 34 s" at 22:02:13Z and paid 0.93116546 IGN at 22:02:24Z. PC 2's prover had been dark from 21:25:24Z to 22:02 (the root-socket class; the CI check tools/ci/prover-socket-check.sh now fails any playbook without the two restore lines). Suite job 3 of 3 (consensus-core, igneum-exec, kaspa-pow, igneum-miner, kaspa-p2p-flows and the app tests; main 8ea6740, fork 79bd8e10) is packing and publishing from the ca2-v3 worktree now. After it on PC 2: the prover-floor agent's toolchain check and its 60-minute build, then the aggregation-cost re-run.
## 22:06 decided: mixer x8 into v3; the verifier fix found; ca2-v3 at 795472e
The mixer PC 1 job (22:00:08 to 22:04:39Z, both cards restored, the 9070 XT present): the daily 1 GiB build per pack, two dispatches, wall ms: RTX 5090 v2 25 / 23, mx4 24 / 24 and 23 / 25, mx8 23 / 23 and 23 / 23 (cache 4); RX 9070 XT v2 74 / 74, mx4 77 / 75 and 73 / 74, mx8 72 / 76 and 75 / 74 (cache 8 to 9). The build is latency-bound on every card; the x8 rule's build half passes with 13x to 40x margin; its verifier half passed at 2.79 ms per warp on the loaded core. DECIDED (delegated): x8 into v3; V3_CLASS = { era: None, hot: None, ..MX8 }; the chip row at x8 reads 0.92x with the 3x factor (the claim "under 1x with the factor", margin stated). Every mixer pack's fingerprint equals the Mac's on both cards (v2 25f96e7dce90bd4e; mx4 6f48d5a2aa0dbe5f, 73caaebb28e808fe; mx8 7c28cfb06c5c65a9, bbb183f72692f840); hash rates at the v2 rate on both (5090 136.5 to 137.4, 9070 XT 18.0 to 18.2 at every class).
The verifier regression is found: not the mask but inlining; the item loop inlined into MemhardCpu::fetch runs at 1.33 ms per unit, the same loop out of line (#[inline(never)], one instance per cache size, the line mask a constant) at 0.60 to 0.62 against readwidth's 0.60 to 0.64 in the same minute. The fix, the MX8 V3_CLASS, the pinned pack re-export and the final v2 / x4 / x8 session land as one commit on ca2-v3 795472e (the node agent merged the mixer's 16dfd1e as 4e733bb with two one-line field fixes; 53 + 4 + 19 + 7 tests, packfile 0 failures, the fork check clean). Then: the node agent merges, rebuilds, gate run 3 on the final class; the six era packs and the pinned v3 pack re-exported on it; the final bit-exactness and G2 (1,000 random hashes per card re-hashed by the CPU) job on PC 1.
## 22:08 G6 job 3 failed on a stale fork test (era inside the class); job 4 on the final tree
build-20261005-220351 (22:04 to 22:06:55Z): igneum-exec 17, igneum-miner 18, consensus-core and p2p-flows green, the app 78 + 26 + 8; kaspa-pow 13 passed, 1 failed: igneum::tests::program_class_v3_seeds_hash_their_own_program_over_their_own_cache asserts the program's class equals V3_CLASS with era: None, but the merged crate carries the drawn era inside the class (left: era Some(EraParams { ... stride_mul 3969900165 ... }); right: era None). A stale fork test written before the era merge, not a behaviour fault; the node agent fixes it on the fork. Correction to the G6 note: the PC's test stage DOES run the v3 engine test, so the PC job is the evidence. Job 4 (the five crates and the app) runs on the final tree once the fork fix and the mixer's verifier fix (with V3_CLASS = MX8) are merged; the publish waits on it. PC 1: the 5090 power sweep has the go (8 min), the AMD sweep may follow on its own presence probe; the final v3 vectors and G2 job (the era agent, 1,000 hashes per card re-hashed on the Mac) is being prepared for about 22:40.
22:09. Fork ca2-v3-node 89dfcb95: the v3 engine test asserts what it meant on the era crate (LoadClass { era: None, ..class } == V3_CLASS and class.era.is_some(); the era bytes equal the seeds'; another era seed keeps the program id, draws another era class, hashes another pow, adds no day cache); Mac `cargo test -p kaspa-pow --features igneum-pow` 14 passed. Main-repo tip 6a705a2. Waiting on the mixer's fix commit for the merge, the rebuild, gate run 3 and suite job 4. PC 2: the prover-floor agent's 3-minute toolchain check has the go; its build waits for job 4.
22:11. PC 2: the prover-floor toolchain check done (floor-toolchain-1, 22:10:03 to 22:10:06Z: nvcc 12.8, cmake 3.28.3, gcc 13.3, clang 18, cargo 1.99.0; no go, so the rebuilt server drops native-gnark, the Groth16 wrap that compressed proofs never use; 16 cores, 30 GB WSL RAM; the live server untouched); its 90-minute build (sm_86, sm_89, sm_120 after C26) is HELD behind suite job 4 (the gate), with a 22:40 fallback: if job 4 is not published by then, the build goes first. Consequences C26 (the arch list, the card, the packaging row, the verify-segment run, "24 GB" kept until the 3060 proves) is with the prover-floor agent; C27 (publish-jobs.sh add runs the prover-socket and bash-body checks and refuses on failure) is being wired by the reviewer's sub-agent, no collision.
## 22:11 the verifier fix and the x8 class are committed (ca2-mixer 1ab8b21); the final code is in
Before / after, the era agent's way (one measure session, 22:07 UTC, readwidth e752fc7's binary beside the fixed one, the same v2 input, load 5.5): readwidth 0.607 / 0.610 ms per unit; the fixed binary 0.609 / 0.611 (was 1.332 / 1.316 on 0fc0ad1). On the fixed binary: x4 1.238 / 1.237 (2.0x v2, worst cold 1.40), x8 2.077 / 2.058 (3.4x, worst cold 2.15), x8 on the devnet seeds 2.058; so the x8 class verifies at 2.1 ms per warp on a loaded core, 4.8x inside the 10 ms gate. Cause and fix: the item loop inlined into MemhardCpu::fetch ran at 2.2x whatever the mask; derive_items_mask #[inline(never)], one instance per cache size (2^26 to 2^30) with the line mask a constant, restores 0.61. V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }; mx8-genesis (program id e323b9dcaf283a6f, fingerprint 7c28cfb06c5c65a9) and mx8-devnet-epoch0 through the chain path with the era inside (class mx8-erad810f22d, program id 73bcbfe8ccf988f1, unit 0 lane 0 d424577fce4a7a60, fingerprint 90f794dd556f7a3b on Metal and Apple OpenCL, 3/3 + 3/3, 96 of 96); the suite 53 + 4 + 19 + 7. chip-model-v3.md headline: x8, 0.92x with the factor, margin 8% on the factor and 9% on the budget; x4 kept as the measured candidate. Owed: the composed mx8-devnet-epoch0's fingerprint on the PCs (in the final-vectors job). Next: the node agent fast-forwards ca2-v3, re-checks, rebuilds, gate run 3; suite job 4 publishes when the tree is settled; the era agent builds the final-vectors and G2 package on 1ab8b21.
22:13. Suite job 4 published: build-20261005-221237 (main d233fa1 = ca2-v3 with the mixer fix merged, V3_CLASS = { era: None, hot: None, ..MX8 }; fork 89dfcb95 with the engine test fixed; the five crates and the app). The node agent's checks, rebuild and gate run 3 on d233fa1 follow; the era agent's final-vectors and G2 package is being built on the final class; PC 1 is the AMD sweep's until about 22:25.
22:15. The era final-class package is ready (zip igneum-ca2-era-pc1b.zip sha256 cb0e9db07e304b11fd4c0591351af46090442ea4f51d60eb86945b96bd28aba3; workers f8d19f0a... and 87647c15... from 1ab8b21 plus the era branch; seven packs on the final class: mx8-devnet-epoch0 and era-0 to era-5, program id 73bcbfe8ccf988f1; Mac 7/7 on Metal and Apple OpenCL, fingerprints equal: mx8-devnet-epoch0 a6752e037514c91a, era-0 64c0ee90bac42624, ..., era-5 43673acc89954d5e). G2 method: one serve-mode job of 1,024 nonces at target ff..ff per card for era-0 and the pinned pack, every nonce a "found g2" line, re-hashed on the Mac with igneum-pow hash-bound --count 1024 (dry run through Apple OpenCL: 1,024 of 1,024 on both packs). Its PC 1 go follows the 5090 power sweep (ahead of the AMD sweep). C24 and C27 closed on branch bash-body-check (7adb1ca, 6805125); the integration merge takes its prover-socket-check.sh over proving-v1's and puts pc1-cpu-prove.ps1 on the allow list.
## 22:16 the 5090 power-limit sweep (PC 1, relay #224, 22:09 to 22:15:37Z); the G2 job has the go
| Cap | Limit W | Draw W | MH/s | MH/W | SM MHz | Memory MHz | Busy |
|---|---|---|---|---|---|---|---|
| 100% | 575 | 316.2 | 115.42 | 0.365 | 3,051 | 13,801 | 92.9% |
| 80% | 460 | 316.1 | 115.60 | 0.366 | 3,050 | 13,801 | 92.5% |
| 65% | 400 (the floor) | 310.6 | 114.46 | 0.369 | 3,051 | 13,801 | 90.1% |
| 50% | 400 (clamped) | 302.4 | 109.20 | 0.361 | 3,050 | 13,801 | 87.5% |
The card draws 302 to 316 W under this program whatever the cap, so a cap above 400 W never binds; the readwidth, era, hot-table and mixer numbers taken at 431 W sit on the flat part of the curve (within 1% of stock); best per watt 65% (400 W) at 0.369 MH/W, a 0.8% hash cost. The cap was restored to 431 W and read back. (The app's 0.3.9 rate of 124 to 141 MH/s in the STATUS lines against 115 here: the API's hash_now sampled every 5 s under the sweep's own load; the bench rows of 136 to 137 MH/s are device time.) "go PC 1" given to the era agent's final-vectors and G2 job at 22:16; the AMD sweep follows it on a fresh probe.
## 22:16 gate G6 GREEN on the final tree
build-20261005-221237 (main d233fa1, fork 89dfcb95, 185 s): every stage ok; kaspa-consensus-core 108 (2 ignored), igneum-exec 17, kaspa-pow 33 + 14 (the v3 engine test, with the igneum-pow feature), igneum-miner 18, kaspa-p2p-flows 7, igneum-app 78 + 26 + 8; 0 failed; with job 2's kaspa-consensus 97 alone, G6 is green. PC 2 goes to the prover-floor agent's 90-minute build (its go at 22:17), then the aggregation-cost re-run, then the prover-floor sweep. Gates: G3 green (the Mac suites on the final class: 53 + 4 + 19 + 7 crate tests, the Metal fuzz, edge, stats and determinism runs, the scratch tests), G4 green (runs 1 and 2; run 3 on the final x8 + era class pending), G6 green; G1 and G2 pending the era agent's PC 1 job (running from 22:16); G5 (the Windows and Mac workers from the same commit) is the ship's build step on the merged tree.
## 22:18 the spec and the public copy carry the final class
docs/spec/01-lottery-hash.md on ca2-coord: 1.8.5 (the mixer x8 form with the measured costs), 1.13.1 (the era draw: stride, interleave, windows, the devnet stand-in, the measured spread), 1.17 (the class v3 vectors), 1.5 (the cache note); earlier tonight 1.12 and 1.13.1 (epoch_len), 1.13.2 (R1 and the emulation rule), 1.13.3 (option C and the step mapping), 1.4.5 and 1.4.6 (generator 3, the class and era in the pack), and 4.3. Public copy: level 1 on the hero and the abstract ("Built for graphics cards. A custom chip gains under 2x, and the model and the bounty are public."), the limits bullet rewritten on the chip row (0.92x with the allowance, approximate; the margin on the numbers page; the next lever named), the level 3 table filled from the final numbers in counter-asic-2-public.md (the bench page section is written from it at the ship). Waiting: the era agent's PC 1 job (G1 on the final class and G2), the node agent's gate run 3; then the integration merge and the ship.
22:19. docs/analysis/proving-methods.md (branch proving-methods e7e0db7, not consensus): top recommendation re-size SP1's own GPU server (the floor is its code: the under-20 GB panic at sp1-gpu builder.rs:37, trace buffers at the maximum shard, a CUDA mempool that never releases), S_p as the dial, one server per card on rigs; the pinned ids stay; fallback and the Apple route: RISC Zero as proof-system version 2 behind the ProofSystem seam (8 GB at po2 19, 16 GB at po2 20, shipped Metal; 3 to 4 agent days); no 12 GB card has run a prover here, so a 4070 or 3060 in the loop is the first action. That branch merges into the 0.3.11 main tree as documentation (no code). evidence.md rows 17, 18 and 19 are rewritten on the measured class v3 (ba8379d).
22:20. The numbers page's Counter ASIC section is written as the bench-log entry "Counter ASIC 2.0, the numbers" (the page is built from the bench log; the litepaper links /bench#counter-asic-2-0-the-numbers); the litepaper's verifier figures moved to the v3 class (2.1 ms per warp on a loaded core, 4.8x inside the gate; the cache 512 MB from year 4). Everything public now carries the final class except the PC rows of the final-class packs and the G2 counts, which the running PC 1 job supplies.
## 22:20 gate G4 run 3 PASS on the final class; the node side is final
Fast-time 3-node network on ca2-v3 d233fa1 (x8 + era + the verifier fix) and fork 89dfcb95, 22:15:04 to 22:19:44Z: every check true; 182 / 122 blocks around the boundary; the v3 ids agree on all three miners; 0 rejected; one sink at 303/303/303; the switch line on 3 of 3; cache ready 179 / 191 ms. Final tips: ca2-v3 5eb2331 (docs only since d233fa1), ca2-v3-node 89dfcb95, both clean; the gate network stopped. Gates: G3 green, G4 green, G6 green; G1 and G2 on the final-class PC 1 job (running since 22:16); G5 at the ship's build step.
## 22:21 gate G4b added: the Mac must mine v3 (the app's --prepare-packs gap)
The node agent's owed item is an app gap, confirmed in the 0.3.11 app branch: engine.rs miner_args pushes --prepare-packs only for non-Metal workers (line 1468 on a223ca9), so the Mac app's Metal worker would get no prepare pack and under class v3 would answer `need` at the first v3 epoch (every Mac stops: the 18:23Z class). Closes in hand: (a) the proving agent adds the flag for every worker (packs/prepare on macOS) with a unit test on miner_args for a Metal card, on the app branch above a223ca9; (b) the node agent runs gate 4: a real Metal miner on this Mac across a v3 boundary on the fast-time network through the miner's --prepare-packs flow. The ship does not go without (a) in the app tree and (b) green; recorded as G4b in the rollout plan's gate list.
22:23. G4b (a) done: app proving-v1 e0de2ab on 5b0d54f: miner_args pushes --prepare-packs for every worker through prepare_packs_arg() (packs\prepare on Windows, packs/prepare elsewhere), the OpenCL-only --job-nonces kept, and start_miner now gives the Metal worker the app data folder as cwd (it had None, which would have broken the relative path: a second latent Mac fault closed); unit test every_worker_gets_the_prepare_directory_in_the_platform_form; 113 + 27 + 8 passed. The 0.3.11 app tree's final code commit is e0de2ab. (b), the Metal gate run, is with the node agent.
22:23. PC 2: floor-build-1 (22:17 to 22:21:12Z) FAILED at cargo exit 101 after 240 s with the error not uploaded (the playbook named the cargo log with a timestamp and could not find it; fixed: a fixed path, the error lines printed on failure); floor-build-2 (the same 90-minute shape) has the go at 22:24; the aggregation-cost re-run follows it. PC 1: the era agent's final-class job runs (from 22:16); the AMD sweep follows on a fresh probe.
## 22:24 gates G1 and G2 GREEN on the final class
Job run-ca2-era-pc1b-20261005 (22:16 to 22:21Z, 304 s, both cards restored, app 0.3.10, the 9070 XT present): the seven final-class packs' 2^24 fingerprints equal on the 5090, the 9070 XT and the M5 Max (mx8-devnet-epoch0 90f794dd556f7a3b; era-0 8e8e070db4eea52d, era-1 891c01b8563bb47e, era-2 e54279fed2831b5d, era-3 77e0ba8abbd0ae62, era-4 d898d8f4f2e7684b, era-5 a6927db380f7efb2); G2: 1,024 of 1,024 per card on era-0 and the pinned pack re-hashed by the Rust verifier. Rates on the final class (MH/s): 5090 135.90 to 137.70 (spread 1.3%), 9070 XT 18.59 to 19.18 (3.1%), M5 Max 27.85 to 27.98 (0.5%); the pinned pack 136.10 / 18.90 / 27.80. Spec 1.17 carries these fingerprints. Gates now: G1, G2, G3, G4, G6 green; G4b (a) done, (b) the Metal gate run pending; G5 at the ship's build. PC 1 goes to the AMD sweep on its presence probe.
22:25. Consequences C29 and D11 applied. C29: the litepaper's verifier line now reads 2.1 ms per warp for class v3 on one M5 Max core at load average 5.5 (the fixed crate; worst cold 2.15), 3.4x the v2 verifier's 0.61 ms, the gate leaving 4.8x (4.6x on the worst cold unit); the "about 1.3 ms quiet" scaling is struck: it came from the slow binary's 2.79 ms session (21:40), and the fixed binary's session (22:07) matched readwidth's quiet v2 figure within 1%, so the fixed numbers are near-quiet. D11: "and the bounty" struck from the hero and the abstract, "standing" from level 1; the copy says "a bounty follows the external review"; the bounty is named only once escrowed (funding.md rule 3; USD 50,000 not funded); Josh decides (D11). The level 3 table and the numbers-page entry now carry the final-class rates (5090 135.90 to 137.70, 9070 XT 18.59 to 19.18, M5 Max 27.85 to 27.98).
## 22:25 the fourth eGPU drop; the era branch final; PC 1 to Ember
The 9070 XT is absent again at 22:24:09Z (relay probe #230: no Sonnet or USB4 router device), the fourth drop today, after serving the hot-table, era (twice) and mixer jobs between 21:31 and 22:21; the AMD clock and power sweep did not start and its rows are owed with this time and reason; the hardware-events table in the rollout plan carries the flapping for the morning. ca2-era is final at 95955c3 on ca2-v3 5eb2331 (six commits: the implementation, the Mac rows, the PC round 1 rows, the G2 hash-bound --count flag, the final-class re-export, the round 2 rows; 53 + 30 tests). "go PC 1 build" given to Ember Tune (its two fetches and the 8-minute Windows app build), then its 8-minute run on the 5090 baseline. PC 2: floor-build-2 running (to about 23:50), then the aggregation-cost re-run, then the prover-floor sweep.
## 22:27 the integration merge, in the node agent's hands; the ship order
A dry run of ca2-v3 into master in a scratch worktree conflicts in docs/bench-log.md (append-only, both kept) and proto-opencl/host.c, where taking ca2-v3's side would drop master's 0.3.10 hunks (the PCI topology and duplicate-platform detection, the select read-back): it must be resolved by hand keeping both. ca2-v3 also lacks readwidth's last three commits (d0018cf the OpenCL scratch local-buffer rule, e752fc7 read-width.md and the overflow fix, 30ff674 the per-watt rows). So the node agent, as ca2-v3's owner, merges readwidth 30ff674 and origin/master (cde561c, 1f0d62c) into ca2-v3 beside gate run 4, re-runs the igneum-pow suite, the packfile test, test-generic.sh on Apple OpenCL and the fork check, and sends the tip. ca2-coord (docs, spec, site, evidence, the plans) stays on its a9e002c base: its code files are untouched readwidth copies, so its merge onto that tip takes ca2-v3's code and brings only the documents (a rebase attempt replayed readwidth's own commits and was aborted). Ship order: master <- ca2-v3' <- ca2-coord <- the tooling and analysis branches (consequences, bash-body-check, amd-prove, card-lifetime, proving-methods, asic-history when it lands), the app branch proving-v1 e0de2ab (on 5b0d54f, in master), the fork ca2-v3-node 89dfcb95 (on 21d4c73c, release-0.3.10's tip). The ASIC-history agent (a202a09dcd24ba1d3) resumed at 22:52 (its clock) in ../igneum-wt-asic-history; its ranked additions are Counter ASIC 3.0's input after the publish.
22:30. C31 applied (a7be43f): the litepaper's "12 GB or more proves full shards" struck (one page, one number: 24 GB); the dataset's step schedule written as the gate 1 proposal on the site and litepaper (D4 is Josh's); the merge rule "take proving-v1's evidence row 16 and its proving sentences" in the rollout plan. PC 1: Ember's build done (build-20261005-222558, 97 s, 6 outputs verified), its tune run on the 5090 has the go. PC 2: floor-build-3 (with the pinned Go toolchain; builds 1 and 2 failed on a missing go, the first unreported by a log-path bug) runs, 25 to 45 minutes expected.
22:32. C32: an app update clears the jobs folder (the 0.3.10 install took PC 1's AMD kit, fetched at 21:23:59Z; the 21:05 reading "cleared by fetch jobs" was wrong), so two rules enter the rollout plan (4a): the 0.3.11 update-now reaches PC 2 only after floor-build-3 closes (PC 2 last on the machine list), and every fetch-then-run pair re-fetches after an update, with a kit presence check at the top of every run playbook.
## 22:33 gate G4b GREEN: the Mac mines v3 end to end; a second Metal outage found and fixed
Gate 4 (22:25 to 22:31Z, a real Metal miner through the miner's --prepare-packs flow on the fast-time network, igneum-bench from ca2-v3 00c55aa): three v3 prepares with the pack dir and the class and era tokens; the worker's v3 prepared lines (252.5 ms the first: program 55.5, dataset 196.9, cache fill 0.9, build 41.1; then 51 and 46 ms with the day resident); every swap "with no pause"; 124 blocks accepted on v3, cpu re-check mismatched 0, need 0, no mismatch or refusal; chain 182 / 123 across the boundary, 0 rejected, one sink. Found and fixed before the run: the Metal worker's serveDataset built every day with the Swift version 2 construction keyed by day only, so a v3 program would have hashed over an x1 dataset and every found would have been refused by the CPU re-check, the same fleet-outage class as the app's missing flag; now a v3 prepare builds the day from the pack's memhard.metal and the store keys datasets by (day, class, era) (00c55aa). Two Mac outages caught by one gate run that the CPU-miner gates could not see; the rule for the next cut: every worker path (Metal, CUDA, OpenCL) mines across a boundary in the gate network before a class change ships. The integration merges (readwidth 30ff674, then origin/master) are in progress on ca2-v3 with the conflicts resolved by hand keeping both sides.
Gates: G1, G2, G3, G4, G4b, G6 GREEN; G5 at the ship's build step. The ship waits only on the merged tip and its checks.
22:33. H = 210,000 lands about 18:45Z on 6 October (DAA 137,041 at 22:32:54Z, 1.002 DAA/s since 15:40Z); the 16:00Z check keeps 2.75 hours. PC 2: floor-build-3 green at 22:32:29Z (240 s on the warm target; the patched sp1-gpu-server 166,768,224 bytes sha256 5568108b..., v6.8.1 c84ada1e with patch 700173fe, sm_86 sm_89 sm_120, the Go tarball's sha256 matched; the live server untouched, miners never stopped); "go PC 2 sweep" given for floor-sweep-1 (9 points, 8 to 10 min) ahead of the aggregation-cost re-run.
## 22:38 the merged tip is in; the 0.3.11 ship is assigned
ca2-v3 fa3c932 (code 49c7e78): readwidth 30ff674 merged at 3566afd (emit.rs two hunks, HEAD's superset; packbench.swift's footprint lines added to the hot-aware RESULT line; bench-log both; read-width.md readwidth's version), origin/master 1f0d62c merged at 49c7e78 (host.c one struct hunk, both fields kept; master's topology, duplicate-platform and select read-back code and ca2-v3's class, era, mh_word, hot and mixer code all auto-merged; packfile.h no conflict); checks all green on 49c7e78: igneum-pow 53 + 4 + 19 + 7, packfile-test 0 failures, the NVRTC emulator test PASS (17 sampled hashes equal to hash-bound), test-generic.sh on Apple OpenCL PASS, the fork's check clean. Fork ca2-v3-node 89dfcb95.
The ship is assigned to the 0.3.10 shipper (ae892a8b0f78fe31c) with every input (rollout plan sections 3, 4, 4a, 7, 8, 8a): release-0.3.11 from origin/master; merges in order ca2-v3 fa3c932, ca2-coord, the app branch proving-v1 (e0de2ab plus the prover log-line commit), bash-body-check e3bd761, consequences, proving-methods e7e0db7, asic-history if it lands; the fork 89dfcb95 under the release tree's vendor/; the override object with every switch; N4 = (DAA at publish + 14,400) rounded up to a multiple of 3,600, N5 = DAA + 14,400; the expected digest c562d70e...; the deadline note "program class v3 + proving v1"; update-now machine by machine with PC 2 last after the prover-floor sweep; the hand nodes, the seed, the digest sweep, the HiveOS package, the merge to master and the push. The node and proving agents stand by for fixes. PC 1: Ember's tune run. PC 2: floor-sweep-1 (from 22:34:11Z), then the aggregation-cost re-run.
## 22:41 the release tree is assembled; counter-asic-3.md written; PC 2 to the aggregation-cost re-run
Release worktree /Users/joshm/Projects/igneum-wt-ship0311, branch release-0.3.11 from origin/master b38f3de: merges in order ca2-v3 fa3c932 (clean), ca2-coord (bench-log both kept), the app branch 22c2363 (bench-log both; docs/evidence.md rows 15 and 16 from proving-v1, 17 to 22 from ca2-coord; the litepaper's schedule sentence from ca2-coord with proving-v1's 24 GB sentence appended), ca2-coord 57844e9 (the nine-field override line), bash-body-check e3bd761 (ci.yml both steps kept, prover-socket-check.sh from bash-body-check), consequences 99fd988, proving-methods e7e0db7, asic-history 9e4af7f; tip b968ee0; the fork worktree vendor/igneum-node-0311 at 89dfcb95 under the release tree. The checks (igneum-pow suite, the app suite, test-generic.sh on Apple OpenCL) run now; then the handoff to the shipper (the coordinator has told it to ship). docs/plans/counter-asic-3.md is written from the ASIC-history agent's seven ranked additions (the partial-store chip and the time-memory curve first, the random daily derivation as a reserve, the mixer cryptanalysis as a genesis gate, the detector and the issuance-triggered bounty, the FPGA lane, the reserve order, the vendor-share metric) with the decisions for Josh.
PC 2: floor-sweep-1 (22:34 to 22:37:55Z, every proof verified by the unpatched host): the shard term is gone but a second floor binds at 12.7 GB measured (10.95 GB the server's own): at Setup the server pre-builds five recursion keys at the fixed 2^27 capacity (3.75 GB) plus the shrink and core keys, 9.7 GB before the first shard; a second patch (every trace buffer sized to its program or shard) follows in about 30 minutes; nothing is under 11.0 GB yet; the public line stays 24 GB. "go PC 2" to the aggregation-cost re-run (20 min); the floor's second build after it; the 0.3.11 update-now reaches PC 2 after both.
## 22:42 the release tree is handed to the shipper; the ship runs
Checks on the merge tip b968ee0 on the Mac: igneum-pow 53 + 4 + 19 + 7, the app 113 + 27 + 8, 0 failed (the OpenCL generic test wants the NVRTC emulator run first; both ran green on the same code at 49c7e78). The coordinator told the shipper directly to ship; it has the tree (its version bump 21173c4 sits on b968ee0 in /Users/joshm/Projects/igneum-wt-ship0311) and I touch nothing in it from here. Handoff message sent with the tip, the fork worktree (vendor/igneum-node-0311 at 89dfcb95), the nine-field override line, the two digests and the machine order (PC 2 last, after the aggregation-cost re-run closing about 23:02 and the prover-floor agent's build 4 and sweep 2 on the v2 patch eda49ab: every trace buffer sized to its padded need, the recursion keys 0.75 to about 0.21 GB each). The shipper reports to the coordinator and me at each step; the node and proving agents stand by.
22:43. PC 1: Ember's tune run (ember-tune-pc1-1) failed at 22:31:06Z after 47 s (the second engine exited at once; no knob touched: the 5090 at its 450 W limit, 2,850 MHz; the 9070 XT present on bus 98 at factory); a collect job reads the exit reason; Ember's re-run comes after PC 1's 0.3.11 update with its fetches republished (C32). Finding from its probe: the AMD helper's gmax range is an offset (-500 to 1000), not MHz; the probe is fixed to keep the clock knob closed on offset ranges and bound the power ladder by plimit_range. The window before PC 1's update goes to the AMD sweep (a fresh kit fetch, the two-read probe, 26 minutes if the card is present); the shipper holds PC 1's update-now until "PC 1 clear" (and PC 2's until "PC 2 clear"); the Mac and the laptop update first. PC 2: the aggregation-cost re-run (to about 23:02), then floor-build-4 and floor-sweep-2.
22:44. The coordinator stopped the AMD telemetry agent at about 23:00Z (its clock; the 9070 XT will not be back on the bus tonight); its AMD sweep rows are owed to the morning (relay/playbooks/amd-card-test.ps1, kit amd-kit.zip); it holds no slot. PC 1 is clear for the 0.3.11 update-now once Ember's collect job closes (the shipper reads the intake for no running job on ae432dc7); PC 2's update still waits on the aggregation-cost re-run and the prover-floor build 4 and sweep 2.
## 22:45 the shipper has the tree; the PC queues for the rollout
The shipper took over release-0.3.11 at cc72f4a (b968ee0 + the bump 21173c4 + one CI commit): three of the tree's own checks had failed and are fixed there (the identity check's "MacBook" pattern matched prose in card-lifetime and proving-methods, reworded; the C32 kit-path check flagged the three tools/proving-v1/pc2-*.ps1 scripts for a bare jobs\ literal, now Test-Path'd; the socket check flagged tools/amd-prove/pc1-cpu-prove.ps1 and its -sp sibling, now allow-listed with the reason); every check green (identity 0 of 220, bash-body 15 of 28, kit-path 14 of 14, socket, markers, workflow shell, relay 17, UI 23). The reviewer's C34 follow-up is with the shipper: packaging/mac/packaged-config.sh line 31 must carry the nine-field object before the DMG and the Windows inputs (a fresh install would otherwise start on the four-field digest). Its order: the Mac node build and the two digest readings, the PC 1 build job (node and app, Linux and Windows), the PC 2 suites from the release worktree, the seed's Linux cross-build, the workers, the app, the DMG, the inputs, the push and CI; then the manifest and the machine order.
PC 1: yours to the shipper now (no running job; Ember's run was "aborted (the app is quitting)" at 22:31:06Z, cause unexplained: no 0.3.11 action existed then; the shipper's PC 1 job output may say). PC 2: agg-cost-pc2-3 runs 22:41:15Z to about 22:59Z (PC 2's app log; the intake lags), then the shipper's suite job, then the prover-floor build 4 and sweep 2, then "PC 2 clear" for the update-now. The AMD telemetry agent is stopped (its rows owed to the morning); Ember's re-run after PC 1 shows 0.3.11 (its engine 054e041, the fetches republished).
22:48. C35: tonight's two unattributed app quits share a shape (PC 2 at 20:01:09Z, 20 s after the efficiency sweep's elevated helper was cancelled; PC 1 at 22:31:06Z, 45 s after Ember's job started a second engine beside the installed app), both killing a measurement in flight, both "quit:" lines naming no source. Ember's re-run is HELD until the cause is named: the Ember owner reads the engine's quit path for every caller (the single-instance lock, the API port bind, the helper protocol, the jobs runner's quit command, the updater) and the line before each quit in both PC logs, gives the quit line its source, and makes the second engine unable to make the installed one quit; one line goes to the 0.3.11 tree through the shipper, else the next cut. The AMD agent's last probe (22:45:01Z): the 9070 XT absent by the PnP list 14 minutes after Ember's ADLX read saw it on bus 98; the link flaps on a scale of minutes; its kit amd-kit-2 is on PC 1 for the morning's window.
## 22:48 the ship's first step report: N4 = N5 = 154,800; the workers built (G5)
Tree release-0.3.11 23bc2b2 (cc72f4a plus the nine-field packaged line). N4 = N5 = 154,800 (DAA 136,967 at 22:45Z at 0.965 blocks/s: the publish near 140,200, tip + 14,400 near 154,600, the first multiple of 3,600 at or above it; the 10,800 floor holds until DAA 144,000, about 00:45Z, else re-pinned). The two Windows workers from this tree: igneum-worker-cuda.exe 2b3b8c92... (1,536,512), igneum-worker-opencl.exe edc4a75d... (478,208), both with the resource block, different from 0.3.10's pair. PC 1's build job build-20261005-224654 (node 89dfcb95, app 23bc2b2, Linux and Windows). The app suite 113 + 27 + 8 and igneum-pow 53 + 4 + 19 + 7 green on the tree. In flight: the fork's Mac node and its two digest readings, the seed's Linux cross-build, the prover build and fixtures; then the inputs, the DMG, the push and CI; PC 2's suite job on "PC 2 suites go" (after the aggregation-cost close at about 22:59).
22:50. Two corrections from the reviewer's log reading, for the morning. (1) PC 2's app quit at 20:01:09Z was an update: its log shows "job update-now-0310 (update-now) starts: 0.3.10 is published: re-read the manifest and install now" at 20:01:04Z, the 49 MB download and the installer start; so a 0.3.10 manifest and an update-now job reached PC 2 at 20:01Z, ninety minutes before the fleet publish at 21:39:59Z; the shipper is asked which publish and jobs file that was and whether it was the same build (an unexplained early publish is a release-process question for the morning). (2) C35's order: "node stopped (exit Some(1))" is written by the engine's stop_node inside the quit path, so on both PCs the node's exit is a consequence of the quit, not its cause; the installed app's Quit comes only from the host's close or /api/quit; for PC 1 at 22:31:06Z the open question is whether Ember's playbook sends /api/quit to the installed app before starting its second engine (the reviewer reads the playbook; Ember's re-run stays held).
22:50. WITHDRAWN, the 22:50 entry's point (1): the update-now-0310 lines on PC 2 are at 21:49:24Z (the fleet update), not 20:01Z; the shipper's "nothing published before 21:33Z" stands, and PC 2's 20:01:09Z quit stays unexplained (a cancelled administrator prompt at 20:00:49Z, a PermissionDenied at 20:00:56Z, then a quit with no source; a next-cut item). C35 narrows to Ember's playbook: relay/playbooks/ember-tune-pc1.ps1 lines 125 to 127 POST api/quit to the URL file in $u when its budget is spent; if $u resolved to the installed app's URL file and the budget check fired at once, the playbook quit the installed app 46 s in. The Ember owner confirms before any re-run; the fix would be that the playbook never addresses the installed app's URL file (its own scratch app dir only) and never calls /api/quit on a URL it did not create.
## 22:52 the digests and the DMG; PC 1's app down since 22:31 (the fleet's hash rate and the ship's PC 1 step)
The shipper's second report (tree 23bc2b2): the two digests on the 0.3.11 Mac node: c562d70e... with no override file (= the node agent's pinned test), 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 with the nine-field object at N4 = N5 = 154,800, the value every node must print after the publish (the node logs the class switch "active from epoch 43" and the proving v1 line); the DMG b7e81d4f... (41,592,041 bytes, the nine fields read back from the image); the seed's Linux node 63cf490d.... Waiting on PC 1's exes (build-20261005-224654) for the inputs, the pin, the push and CI.
PC 1 (Ember's finding, confirmed on the intake at 22:51): the installed app has not come back since its quit at 22:31:06Z (the last upload 22:31:08Z, no new run id, no job since), so PC 1 is not mining (the devnet short its 141 MH/s for 20 minutes), the shipper's build job cannot start there, and no update-now can reach it. The relay agent on PC 1 is alive (read-only probes ran through it at 22:45); the shipper is asked to relaunch the installed app through a relay task in the interactive session and to verify a new run id; if the relay cannot reach the user's session, PC 1 waits for Josh in the morning and the 0.3.11 rollout goes without it (its update lands at its relaunch). Josh is not woken. The quit's cause (C35): the senders are the tray Quit, stdin EOF in wrapper mode and POST /api/quit with the token; Ember's playbook POSTs api/quit to the URL file in $u when its budget is spent (lines 125 to 127); the Ember owner is checking what $u resolved to at 22:31; Ember's re-run stays held, and the playbook rule becomes: never read the installed app's URL file, never POST quit to a URL it did not create.
22:53. PC 1: the relaunch of the installed app went out through the relay at 22:52:31Z (item #241: igneum-app.exe started through explorer.exe so the app gets the user's own token, a one-shot scheduled task at limited run level as the fallback); the shipper reports the new run id and the first STATUS line; its build job starts when the app fetches the jobs file. PC 2 queue after the aggregation-cost job (closing about 22:59): the shipper's 0.3.11 suites, the prover-floor build 4 and sweep 2, the aggregation-cost re-run (20 min, with a plain-text parser: under 0.3.10 the app's /api/state comes back empty to PowerShell 5.1's JSON reader while the body is there, a class hit by three playbooks tonight; the app owner fixes the response shape or every playbook parses the text), then "PC 2 clear" for the update-now, then the ledger-pc2 agent's M16 / E17 / P17 job (the inline-cache kernel at 64 and 256 MiB on the 5090 against the honest kernel, the nvidia-smi line per setting, the igneum-exec suite in WSL2; about 12 minutes, after its kit lands on the dl folder in 1 to 2 hours).
22:54. C35 resolution (Ember owner): $u resolved to %LOCALAPPDATA%\igneum-tune\app\app.url, the second engine's own file (lines 34 to 36 and 125 of the playbook; line 74 deletes it before the engine starts; platform::data_root() honours IGNEUM_APP_DATA, set to the scratch root at line 93), and the budget branch never ran (its "RESULT TUNE error=budget_exceeded" line is absent); so the playbook did not quit the installed app. What the reading did find: the second engine counted --sweep as Power control and raised a UAC prompt at about 22:30:25Z (apply_power_limits at start), 40 s before the installed app's quit; closed on ember-tune b671c8b (Power control alone decides, no cap at start under --sweep, every Cmd::Quit names its source, the budget floored at 5 minutes, the playbook refuses a quit to any URL under the installed igneum\app). The remaining question is whether an unanswered elevation prompt can take the installed app's window host down (stdin EOF): the same shape as PC 2's quit at 20:01:09Z, 20 s after a cancelled administrator prompt (C16). "go PC 1 collect" given: the Application event log at 22:31Z and the installed app's log tail, once PC 1's app is back; the re-run stays held until the source is named.
22:55. C35 narrows to one common factor: both unexplained quits came 20 to 41 s after an administrator prompt beside the running installed app (PC 2 at 20:00:49Z then 20:01:09Z; PC 1 at about 22:30:25Z then 22:31:06Z); Ember's playbook is ruled out. Rule for the night (rollout plan 4a): no PC job raises an elevation prompt on either PC; the two-minute class test (one prompt raised and cancelled beside the mining app, the quit line read) waits for the morning with Josh present, or tonight only after the relay relaunch path is proven and the rollout is done; Ember's quit-source stamping (b671c8b) goes to the next cut. The devnet has been short PC 1's 141 MH/s since 22:31Z (96 MH/s at 22:51); the relay relaunch #241 is the recovery, else PC 1 is on the morning's hands list beside the eGPU reseat and the 16:00Z check.
22:55. PC 1's engine did not exit: relay task #241 found igneum-app.exe ALIVE (pid 26696, started 21:49:40Z, session 1) answering nothing on /api/state in 60 s and uploading nothing since 22:31:08Z: it logged its quit at 22:31:06Z, stopped the miners and the node, and hung instead of exiting (a quit that never ends: a C35 fact and a next-cut defect: the quit path must end the process or the watchdog must end it after a bound). Task #243 (22:55:10Z) ends that engine by pid, as the app's own updater does, then starts the per-user install through explorer.exe (the user's token), the limited-run-level scheduled task as the fallback; the new run id, the first STATUS line and the build job's first STAGE line follow in the intake.
22:56. fud-close (the ledger closer's 45 public-text fixes, two CI checks, the relay fixes; 0 conflicts with ca2-coord) is added to the ship order after ca2-coord if its ready tip reaches the shipper before the inputs are pushed (the workers and the DMG rebuilt from the merged tip, G5); else it heads the next cut. The fork-side ledger-fixes (from release-0.3.6) is not in 0.3.11.
22:57. Correction: fud-close is NOT in 0.3.11 (the tree closed at 23bc2b2 before it reached the ship order; no late branch, the 0.3.10 rule); it heads the next cut with the fork-side ledger-fixes rebased onto the 0.3.11 fork. The push and CI go the moment PC 1's exes land.
22:57. Next-cut coupling recorded (the ledger closer): fud-close 647b08c's worker half of M28 (the kernel_sha256 check in packfile.h and host.c) must ship with the fork-side ledger-fixes miner commit 3d4ec451 that stamps the hashes, or every worker refuses every pack; ledger-fixes is not yet rebased onto 89dfcb95 (two conflicting files); round 2's ledger branches stay behind tonight.
22:59. Next-cut note from the reviewer's merge-tree against 23bc2b2 (0 conflicts for explorer d7e797c, ember-tune b671c8b, rig-install 086008a, pool-v0 425b875, ota-k2 89a76b1, hive-words 2d056e8, fud-close 647b08c and the others already in): explorer d7e797c makes tools/ci/public-api-check.mjs fail when the live /api/stats lacks `proving`, and that check runs on master pushes against the live site, which Vercel redeploys only after the push, so the first master CI after a ship carrying it goes red through no fault; the next cut holds d7e797c or gives the check a retry loop. Waiting now on two watchers: the aggregation-cost close on PC 2 (then the shipper's suites) and PC 1's app relaunch (then the exes, the push and CI).
## 22:59 PC 1: the hang explained; orphaned miners from the second engine hold both GPUs
The shipper's relay task #244 (22:55:24Z) counted 1 igneumd, 2 igneum-miner, 2 igneum-worker-cuda and 1 igneum-worker-opencl running although the installed app had stopped its miners at 22:30:20Z and its node at 22:31:06Z: Ember's second engine's children, orphaned when its job was aborted, mining on both GPUs; they would fight the relaunched app's miners and void every number. The relay lane kills the tree (taskkill /F /T on every miner, worker and non-app igneumd) and relaunches the app. The hang (Ember's reading): the installed engine's quit got stuck in the jobs runner's abort, whose reader waits for EOF on the script's stdout pipe; the pipe's write end was inherited by the second engine and its miners (PowerShell's Process.Start with redirection inherits every inheritable handle), so EOF never came and the engine sat "responding" until #243 ended it. Class rule for every playbook that starts a second engine (ember-tune-pc1.ps1, relay/playbooks/sweep-5090.ps1 and any job script of the shipper's): no inherited pipe into the second engine, its whole process tree killed at the end and on abort, the installed app's miners restarted only after; a CI check that fails a playbook starting an engine without those lines (Ember's branch). The quit's sender is still open (the tray excluded by the missing 45-s host timer kill: stdin EOF or POST /api/quit; the event-log collect decides, when the app is back).
23:00. The second-engine rule is closed on ember-tune 8ab9068 (docs/plans/ember-tune.md section 5; the playbook pair fixed; tools/ci/second-engine-check.sh in ci.yml, shown to fire on a known-bad playbook and pass the fixed pair); next cut. The shipper's relay kill-and-relaunch task on PC 1 goes out at about 23:03:30Z (or the kill alone at once if the new engine reports first); the build job follows on a clean PC 1.
23:03. C38 (the reviewer): the documents of ca2-analysis ee42d7c (sram-mirror.md, int8-matrix-family.md, the dot4 probe sources), ca2-soundness a465881 (scratch-soundness.md; its scratch tests reached ca2-v3 through the mixer branch), ca2-epoch e95e8b5 (epoch-length.md) and prover-floor (prover-floor.md) exist on their branches only: my ship list omitted them, and evidence rows 17 and 19, the litepaper's chip bullet, chip-model-v3.md and the rollout plan's layer 9 row cite them. The shipper decides before the push: a docs-plus-standalone-sources merge of the four (0 conflicts for docs/, nothing the gates ran on, the shipped code paths' diff verified empty), or, under the 0.3.10 no-late-branch rule, the citations changed to "on branch <name>" on ca2-coord as one docs commit and the four heading the next cut beside fud-close.
23:05. C38 closed on ca2-coord: ca2-analysis ee42d7c, ca2-soundness a465881, ca2-epoch e95e8b5 and prover-floor cfe3d80 merged (8c00f84, 271cd63, f940101, ead67e0; the bench log both sides each time), the soundness branch's two code files set to the release tree's versions (0b505f9; an empty diff against 23bc2b2), so the five cited documents are on the branch the ship takes and its merge is one docs commit; the shipper decides whether at 0.3.11's master merge or the morning's cut.

View file

@ -16,6 +16,21 @@ Trigger: the 9070 XT measurement of 5 October (bench-log "the 9070 XT on the eGP
| 6 | Cache growth on the genesis schedule | the SRAM mirror stays unaffordable | none | already in the design; confirm the schedule against SRAM density |
| 7 | Integer matrix ops in the program (INT8 x INT8 into INT32, exact) | matrix hardware at GPU scale | none on NVIDIA and AMD; Apple to check | reserved family, not at launch |
| 8 | Working-set size drawn per program | one memory design cannot fit every hour | none | folds into 4 and 5 |
| 9 | Epoch length as a signalled era parameter: base 3,600 DAA s, ladder 600 to 7,200, set by 90% signal at a day boundary, lead and `T_epoch` fixed | a per-program bitstream (an FPGA with a hard datapath): at 600 s nothing it compiles ever runs (42 to 160 min per compile, PRflow FPT 2019; hours on large parts, Aldec) | compile-ahead 1 s per epoch on the 5090, 0.5 s on the M5 Max (38 s with the race on); one CPU core `600 / epoch_len` busy on the VDF | reserve-only tonight: `docs/plans/epoch-length.md` |
## Decided 5 October 2026 (night), under Josh's delegation for the devnet (`docs/plans/counter-asic-2-rollout.md` section 6)
| # | Decision | The number that decided it |
|---|---|---|
| 1 | keep v2's 128 x 4 B | w16 passes the rule but closes nothing (5090 139.8 against 136.1 MH/s, 9070 XT 17.90 against 18.15); w64 makes the 5090 bandwidth-bound (share 0.58, 37% of stream) |
| 2 | out | spread across six programs 5.5 to 22.3% per card, over the 5% rule |
| 3 | out (scratch share 0); the construct is sound and its tests stay | the on-die-cache recompute chip stays at 2.4x at every share under the 6 GB cap |
| 4 and 8 | IN: the era draw of stride, interleave and the working-set window (width pinned at 4 B) | six-era spread 1.3% on the 5090, 3.2% on the 9070 XT, 0.8% on the M5 Max; bit-exact on all three vendors |
| 5 | OUT of v3 (a measured option for 3.0) | the added form costs the 5090 13 to 16% and the 9070 XT 16 to 20% (g 0.87 / 0.85 / 0.84 and 0.84 / 0.81 / 0.80 at 32 / 64 / 96 MiB against the 0.97 rule): no card keeps even 32 MiB resident while the dataset streams; the replaced form helps the on-die-cache chip (x1.33 at k = 4) |
| 6 | option C: the cache doubles when the dataset doubles | the mirror is 128 mm^2 and $46 at N5 by shipped density; the cache's job is to stay above GPU L2 |
| 7 | reserve R1 = mm8, unsigned, W_new 4, unlock era 4 or 90% signal | native on all three vendors as a tile; dot4 emulation 1.6x on Apple |
| M16 mixer x8 (x4 measured beside it) | into v3 | the only measured lever that moves the named chip: x8 0.31x bare, 0.92x with a 3x fixed-function factor (x4: 0.61x, 1.84x); verifier 2.1x v2 per warp (2.79 ms on a loaded core, about 1.3 ms quiet); the daily build unmoved on every card (latency-bound) |
| 9 | reserve-only, no change to the devnet's hour | the floor 600 s from the slowest compile-ahead (38 s) |
Not added: divergent data-dependent branches (cost GPUs more than chips), anything floating point (bit-exactness across vendors).

View file

@ -0,0 +1,30 @@
# Counter ASIC 3.0
The third set of chip-resistance layers, from the ASIC-history agent's audit of 5 October 2026 (`docs/analysis/asic-resistance-history.md`, branch asic-history: 31 chip histories with the gain per joule, the months held and the response; the chip economics; the audit of class v2 and of every Counter ASIC 2.0 layer). The rule is the 2.0 rule: every item is measured the same way (the three cards we own, the chip model per variant, bit-exactness, the verifier cost), what passes is folded into the class as v4 behind its own activation (`program_class_v4_activation_daa`, by DAA height like v3), under the same six gates and the same rollout shape (`docs/plans/counter-asic-2-rollout.md`). Nothing here is active; nothing touches the devnet until it has its measurement and Josh's word. Written at 22:4x UTC on 5 October 2026, after the 2.0 class was decided and before its publish.
## The ranked additions
| # | Addition | What it takes from a chip | Where it sits | Step |
|---|---|---|---|---|
| 1 | The partial-store chip and the time-memory curve: price a chip that holds a fraction f of the dataset (f = 0.25, 0.5, 1) on HBM3 or GDDR7 with 4-byte access granularity and recomputes the rest, scored in energy per hash | the only chip class that beat a memory-bound GPU hash (Ethash: 2.1x Linzhi 2020, 2.9x E9 2022, 4.8x per joule Jasminer X4 2021) did it with custom memory controllers and on-package memory, not an on-die dataset; chip-model-v3.md prices only f = 0 | `docs/analysis/chip-model-v3.md`, O-1.6, MEMHARD.md section 3 item 2 (the curve never drawn) | analysis, before the public testnet's vectors freeze; the first item |
| 2 | A random item-derivation program per day in place of the fixed-shape mixer (RandomX's SuperscalarHash idea) | the fixed mixer shape IS the 3x fixed-function allowance that turns x8's 0.31x into 0.92x; removing it is worth more than x16 (0.46x with the factor) | a reserve family now; genesis if the per-day compiled derivation verifies under the 10 ms gate (unmeasured); risks: cryptanalysis of random ARX, weak draws, bit-exact compilation on three vendors; the daily build about doubles (23 to 77 ms, approximate) | design and the verifier measurement |
| 3 | External cryptanalysis of the mixer M_r, the chained cache and the acceptance rule, with the x8 shape as the target | MTP fell from 2 GB to under 1 MB before launch (Dinur and Nadler 2017), Catena's proofs were flawed, Argon2i's parameters were attackable; RandomX bought four audits for about $141,000 before launch; x8 multiplies the mixer's weight in the chip model, so a structural shortcut is worth 8x more | ledger M7, raised to a genesis gate | commission before genesis |
| 4 | The clock and the detector: (a) a share-pattern detector on the observer (per-program hash-rate spread, nonce-group patterns, per-card-model rate bands; alert when a population behaves like one fixed design: how MoneroCrusher found Monero's secret chips at 85% of the hashrate, February 2019); (b) the bounty's trigger as daily issuance in dollars, not a date (compute-bound hashes got chips at $20K to $30K a day: Radiant, Kadena, Handshake; Vorick's 2018 rule about $55K a day) | not a layer: the response time | the observer (`tools/observer`), D11 (the bounty is unfunded) | before the public testnet |
| 5 | Rank layer 9 (the epoch length) above layer 7 and measure the FPGA lane: a soft-overlay FPGA with HBM (reads in flight per watt against the 5090's 17.5 G/s) added to the compile-ahead measurement | FPGAs were the first adversary of Lyra2REv2 (2018) and X16R (1.3x, September 2019) and came back within weeks of X16Rv2; Xelis forked for FPGA resistance (July 2024); a per-hour compiled program is a bitstream target | `docs/plans/epoch-length.md` | measurement before the public testnet |
| 6 | Order the reserve by chip-unfriendliness: the 32-bit datapath families first (byte permute, bit-field extract, variable shifts, popcount, select, the second shuffle), mm8 last | int8 matrix blocks are licensable IP at every node; Apple pays 1.6x to 4.7x per emulated dot4; Least Authority's ProgPoW suggestion 5 was "watch ML hardware" | spec 1.13.2 | a decision for Josh with the 3.0 measurements |
| 7 | A vendor-share metric (hashrate by vendor) published with the benchmark | the 7.5x AMD gap is a softer form of the capture the history records (Kaspa's GPU share went to nothing within months of KS0) | the numbers page, the observer | with the public benchmark |
Placed nowhere, with the reasons in the history document's section 4.3: per-hash programs (the 25x GPU penalty RandomX pays), Verthash's table-from-chain, Grin's dual PoW, Autolykos non-outsourceability, a per-hash VRF (NexaPow's got a 3x to 4x chip), ternary or variable-precision ops, cache-timing reads (measured out as layers 3 and 5), branches and floating point (excluded).
## Decisions for Josh raised by the history
Add the partial-store rows before the vectors freeze; name the random derivation as a reserve family and fund its verifier measurement; commission the mixer cryptanalysis; escrow the bounty on an issuance trigger and build the detector; rank the epoch-length reserve above mm8.
## The plan, in order
1. Item 1 (analysis): the partial-store chip rows and the time-memory curve in `docs/analysis/chip-model-v3.md`, with the sector size per card (the 5090 moves a 32-byte sector per 4-byte read, the 9070 XT 64) and the HBM3 and GDDR7 random-read rates cited; the first item because it can move the public claim.
2. Item 2 (design and measurement): the per-day derivation drawn from a reviewed fixed set, the verifier cost on one core, the daily build on the three cards; reserve entry text for 1.13.2.
3. Items 4 and 5 (tooling and measurement): the detector on the observer, the issuance trigger, the FPGA overlay estimate.
4. Item 3 (procurement): the cryptanalysis brief and the target shape.
5. Items 6 and 7 (decisions and the benchmark page).
6. What passes, as class v4 behind `program_class_v4_activation_daa`, the six gates, the rollout shape of 2.0.

235
docs/plans/epoch-length.md Normal file
View file

@ -0,0 +1,235 @@
# Epoch length as an era parameter (Counter ASIC 2.0, layer 9)
5 October 2026 (night), branch `ca2-epoch`, worker "ca2-epoch". Josh: "what about faster program changes?". Status: Designed, with one Measured section (the Mac compile-ahead, section 6) and cited figures for the PC cards. Reserve-only tonight: nothing here changes the devnet's 3,600-DAA-s epoch, no code moves, no consensus effect. The parameter joins the era-parameter table of spec 01 section 1.13.1 beside the layer 4 and 8 draws of `docs/plans/era-layout.md` (branch `ca2-era`).
## 1. What the layer is
Today the program changes every 3,600 DAA s (spec 01 section 1.12, "a prototype value, to be fixed at gate 2"). This layer makes the length a genesis-reserved parameter, `epoch_len`, base 3,600, settable by 90% miner signal (spec 05 section 5.7) on a fixed ladder between 600 and 7,200 DAA s, with no fork and no release. The point is a chip that must be rebuilt per program (an FPGA with a hard datapath): the shorter the epoch, the smaller the share of each epoch it can mine. Against a GPU the cost is the compile-ahead cadence, measured in section 6, which is what decides the floor.
| Decision | Choice | Why |
|---|---|---|
| How the length is set | 90% signal only; the era stream consumes one draw for it and ignores the value (section 2.4) | a random length buys nothing against the threat and moves two costs (difficulty settle, VDF duty) at random every era |
| Base | 3,600 DAA s | the devnet's hour, unchanged |
| Ladder | 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 (index 0 to 7) | every step divides the day (86,400) and the era (15,552,000), so day and era boundaries are epoch boundaries at every length |
| Where the schedule anchors | the day, not the era (section 2.2) | a change lands at a day boundary, days after the signal, not 180 days after |
| The seed lead | unchanged: checkpoint at least 1,200 DAA s before the epoch start, `T_epoch` fixed at 600 s on the reference core (section 3) | the known-program window stays 600 s at every length; the grinding margin and the checkpoint depth are untouched |
| The floor | 600 DAA s (section 6.3) | the slowest measured compile-ahead (the Metal race, 38 s) is 6.3% of 600 s and inside the 600-s window |
## 2. The parameter
### 2.1 Definition
`epoch_len(d)` is the epoch length in DAA seconds in force on day `d` (spec 01 section 1.12: day `d` covers `[86,400 d, 86,400 (d + 1))`). Epoch `(d, e)` covers DAA scores `[86,400 d + epoch_len(d) e, 86,400 d + epoch_len(d) (e + 1))` for `e` in `0 .. 86,400 / epoch_len(d)`. Every ladder step divides 86,400, so the last epoch of a day ends exactly at the day boundary and the first epoch of a day starts at it, as 1.12's day key already assumes ("the program seed of the first epoch of day d").
The identifier of an epoch is its start score `s = 86,400 d + epoch_len(d) e`, not an index. Everything that today takes the epoch index `e` (the seed checkpoint rule of spec 04 section 4.3 step 1, `seed_source` in the header, the hot table key of layer 5, the program id) takes `s` instead. At the base length `s = 3,600 e`, so nothing changes for the devnet.
"Which program was this block mined under" stays a function of the header alone: the header's DAA score gives `d`; `epoch_len(d)` is a function of the chain state before day `d - 2` (section 2.3), which is in the header's own past, exactly as the era parameters `E_n` are; so `s` follows, then `C(s)`, then the seed. Two nodes validating the same header derive the same `s`.
### 2.2 Why the day and not the era
Anchoring to the era would make a change wait up to 180 days, which defeats the purpose ("miners can shorten it when an FPGA appears"). Anchoring to the day makes the response time the signalling window plus two days. The cost is one more thing the day boundary does; it already resets the day key, the cache and the dataset, so the program change at a day boundary is already paid. The parameter still lives in the era table of 1.13.1 (its base, bound and draw slot are genesis constants there); only its activation clock is the day.
### 2.3 The signal
Spec 05 section 5.7's mechanism, with its open items (window length, delay, field) answered here for this parameter only:
| Item | Value | Note |
|---|---|---|
| Field | 3 bits of the header `version` (Kaspa's version-bits field), the ladder index 0 to 7; 0 to 7 all valid, the current index is the default a miner carries | the proposal-id format of O-5.3 stays open for code upgrades; this is a parameter vote, not a code upgrade |
| Window | 7 days of blue blocks (604,800 at 1 block/s, approximate: counted in blocks, not time) | long enough that a rental burst cannot swing it; short enough to answer an FPGA in a week |
| Threshold | 90% of blue blocks in the window carry the same index, and it differs from the index in force | the 90% rule of 5.7; a split vote changes nothing |
| Activation | the first day boundary at least 2 days after the window closes | 2 days covers the 1,200-s seed lead and every compile-ahead measured in section 6 many times over; a node sees the change coming 2 days ahead |
| Hysteresis | none beyond the 90%: to move again, 90% must carry a new index in a fresh window | |
`epoch_len(d)` is therefore: the base (3,600) at genesis; after each activation, the activated index's length. A node derives it from the blue blocks of the window, which are in the header's past.
### 2.4 The draw slot
The era stream of `era-layout.md` section 1.1 draws seven values in a fixed order. This parameter takes draw 8: `next()` is consumed and the value is not used. Reason to consume it: the stream layout is fixed at genesis; if a draw is ever wanted for this parameter it changes no other parameter's value. Reason not to use it: a drawn length answers no threat (an FPGA fleet is defeated by the minimum, not by the variance), and it would move the difficulty-settle share (section 4) and the VDF duty (section 3.3) at random every 180 days.
### 2.5 The genesis reserve row (spec 01 section 1.13.1)
| Parameter | Base (era 0) | Draw | Bound |
|---|---|---|---|
| Epoch length `epoch_len` | 3,600 DAA s | draw 8 of the era stream consumed, not used; set by 90% signal (section 2.3) at a day boundary | the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 |
Unlock condition: live at genesis at the base; the signal path is the unlock. Nothing moves without 90% of blue blocks over 7 days. No height gate is needed because the base is today's value.
## 3. The seed path at a shorter epoch
### 3.1 Two options for the VDF
Spec 04 section 4.3: the seed of epoch `s` is the 10-minute VDF (`T_epoch`, 600 s on the reference core) of the checkpoint at least 1,200 DAA s before `s`. The output is known to a reference core 600 s before the epoch starts (a core half as fast finishes at the boundary). So the usable compile-ahead window is 600 s, at every length.
| Option | Rule | Known-program window | Grinding margin (spec 04 section 4.6, against the 2-s window) | Checkpoint depth | VDF duty per core | Verdict |
|---|---|---|---|---|---|---|
| A | `T_epoch` and the 1,200-s lead stay genesis constants whatever `epoch_len` | 600 s at every length (lead minus evaluation) | 300x at every length | 1,200 s plus d at every length: a reorg across it stays a merge-depth-scale event (merge depth 3,600 s, spec 04 section 4.3) | `600 / epoch_len`: 17% at 3,600, 33% at 1,800, 100% at 600 | RECOMMENDED |
| B | `T = epoch_len / 6`, lead `= epoch_len / 3` | `epoch_len / 6`: 100 s at 600 | `epoch_len / 6 / 2 s`: 50x at 600 | `epoch_len / 3`: 200 s at 600, a tenth of the merge depth, so a reorg across the seed checkpoint is an ordinary event | 17% at every length | not recommended: the seed checkpoint gets shallow and the margin falls 6x at the floor |
A at the floor: the checkpoint for epoch `s` is at `s - 1,200`, two epochs back; the VDF finishes at `s - 600`, the start of the previous epoch; so each epoch's program is known for exactly one epoch before it starts. "The seed input for `s + 1,200` is known during epoch `s`" is true but useless to a grinder or a compiler: the input does not give the program before the VDF finishes.
### 3.2 Grinding (proto-vdf/README.md, grinding table)
The table's model is per epoch: a favourable program is worth `3,600 s (1 + a) / (1 + s a)` blocks of revenue for the epoch, against one burned block. The benefit is proportional to the epoch length and the cost is not, so without the delay the gain-to-cost ratio falls in proportion: 6.0 to 15.1 : 1 at 3,600 becomes 1.0 to 2.5 : 1 at 600 (the same model, scaled; not re-run) and 12 to 30 : 1 at 7,200. With the delay the gain is 0 at every length, because the grinder learns nothing inside the 2-s window; option A keeps the 300x margin that makes that true. A shorter epoch weakens the attacker's prize and leaves the defence as it is; a longer one raises the prize and the defence still holds at 300x.
### 3.3 The VDF duty (a consequence, every tier)
Under A a node evaluates a 600-s VDF once per epoch, so one core is busy `600 / epoch_len` of the time: 8% at 7,200, 17% at 3,600, 33% at 1,800, 50% at 1,200, 100% at 600. At the floor every mining node spends one core on the VDF without pause (the M5 Max core at 163,000 squarings/s, spec 04 section 4.2; x86 with AVX-512 IFMA faster, open item there). What that means per tier: a pool user, nothing (the pool evaluates); a home miner on a 4-core box, a quarter of the CPU at the floor and the option of spec 04 section 4.7 B (take `(y, pi)` from a peer, verify in 4.5 ms) if it has no core to spare; a rig, one core for all its cards. Proof traffic: 516 bytes per epoch, 6x per hour at the floor (3 KB an hour); verification 4.5 ms per epoch. The floor is an emergency setting for the day an FPGA appears; at the base nothing changes.
### 3.4 Fallback (spec 04 section 4.7)
Unchanged. Option B there (no consensus fallback, proofs from peers) matters more at the floor, because a node slower than 2x the reference core and isolated loses up to a whole epoch instead of a sixth of one. Decision at gate 3 as written.
## 4. The difficulty window as a constraint on the floor
Spec 02: rule v2's reference lane is the newest `REF_WINDOW_V2 = 600` DAA s of the epoch, the short lane is 120 chain blocks, and the lanes are epoch-bounded because the hash rate steps by program (35 to 48 Mhash/s across seeds on the M5 Max, spec 01 section 1.12). The constraint:
| Quantity | Rule | At 600 | At 1,800 | At 3,600 | At 7,200 |
|---|---|---|---|---|---|
| `REF_WINDOW_V2` | `min(600, epoch_len)`; the lane never spans a program change | 600 (the whole epoch) | 600 | 600 | 600 |
| Settle after an epoch step, measured | 144 s on the +-30% epoch-step scenario (simulator, `docs/bench-log.md` 3 October difficulty entry; the live v2 settled a 1.85x in-epoch step within 10 minutes, `docs/analysis/difficulty-2026-10-04-oscillation.md`) | 24% of the epoch | 8% | 4% | 2% |
| Rate error during the settle | bounded by the step (the program-to-program spread, +-15% on the M5 Max; the sign is random per program) | | | | |
At the floor the reference lane is the whole epoch and the v2 cap does nothing; the short lane tracks, as the measurement says it does within 10 minutes. The cost is the settle share: with the measured 144 s a 600-s epoch spends about a quarter of its blocks re-settling after each program change, at a rate error up to the program step. Emission follows the block rate (spec 02: `E` per block), so emission wobbles by up to the step for 144 s per boundary; the sign is random per program, so the drift averages near zero and the variance rises. The 144 s is the +-30% scenario; the +-15% program step is not measured (owed, section 9: `sim/difficulty/sim.py` has `EPOCH = 3600` as a constant). Under the under-10% rule applied to this cost the number lands at 1,800 s, which is why the ladder has steps between the floor and the base: the signal can stop at 1,800 and the floor stays reserved for the case that needs it.
## 5. The threat it answers, and the one it does not
### 5.1 An FPGA fleet with a hard datapath per program
A program is 64 instructions x 8 iterations with 16 loads per iteration from a 1 GiB table (spec 01). A fleet that maps each hour's program to a fixed datapath must synthesise, place and route a bitstream per program and load it, inside the 600-s window of section 3.1. Published compile times:
| Source | Device | Figure |
|---|---|---|
| Xiao et al., "Reducing FPGA Compile Time with Separate Compilation for FPGA Building Blocks" (PRflow), FPT 2019, University of Pennsylvania, https://ic.ese.upenn.edu/pdf/prflow_fpt2019.pdf | ZCU102 (XCZU9EG, about 600 k logic cells, approximate) | monolithic Vivado compile of their benchmarks 42 minutes typical, one case 160 minutes; "hour-long compilation times"; their partitioned flow 12 to 18 minutes |
| Aldec, "Save hours of Place & Route time", https://www.aldec.com/en/company/blog/92--save-hours-of-place-and-route-time-in-seconds | Virtex UltraScale class (4.4 M logic cells) | place and route in hours for large designs; UG904's incremental flow about 3x faster when at least 95% of cells and nets are unchanged |
| AMD UG904, Vivado Implementation User Guide (cited through the Aldec blog and the Vivado documentation) | | place and route is the longest stage; incremental reuse needs a 95% similar design, which a fresh random program is not |
The window is 600 s at every ladder step (section 3.1); no row above fits it, so a per-program bitstream can never mine the start of an epoch. What the epoch length sets is the share of each epoch the fleet mines once the bitstream lands: `max(0, 1 - (compile - 600) / epoch_len)`.
| Compile time per program | Share of each epoch mined at 600 | at 1,800 | at 3,600 | at 7,200 |
|---|---|---|---|---|
| 12 min (PRflow partitioned, a small kernel) | 0% | 93% | 97% | 98% |
| 42 min (PRflow monolithic) | 0% | 0% | 47% | 74% |
| 160 min (PRflow worst) | 0% | 0% | 0% | 0% |
| hours (Aldec, large parts) | 0% | 0% | 0% | 0% |
Reading: at the base an FPGA fleet with a fast compile farm mines half to all of each hour; at 7,200 a small kernel is a non-issue for it; at 600 nothing it compiles ever runs. That is the lever the signal gives miners. A compile farm does not help a fleet: the window is wall time per program, not throughput, and the same window applies to every board.
### 5.2 Partial reconfiguration
Loading a partial bitstream is fast: the ICAP takes 4 bytes per cycle at 100 MHz, 400 MB/s, so a region reloads in milliseconds (Microsoft Research, "Minimizing Partial Reconfiguration Overhead with Fully Streaming DMA Engines and Intelligent ICAP Controller", measured 399.6 MB/s, https://www.microsoft.com/en-us/research/wp-content/uploads/2016/02/Minimizing20Partial20Reconfiguration20Overhead20with20Fully20Streaming20DMA20Engines20and20Intelligent20ICAP20Controller.pdf; AMD UG909, Dynamic Function eXchange, https://www.amd.com/content/dam/xilinx/support/documents/sw_manuals/xilinx2021_1/ug909-vivado-partial-reconfiguration.pdf). It does not shorten the compile: the partial bitstream is placed and routed by the same flow first. A smaller region compiles faster than the whole part (the PRflow rows above are that effect), which is why the 12-minute row is in the table; the 600-s window still beats it.
### 5.3 What the layer does not answer
A programmable chip: a soft processor or a GPU-like overlay on an FPGA, or a custom chip with a programmable core, runs any program with no synthesis, and the epoch length does nothing to it. Those are answered by the other layers (the latency bound, the cache and dataset sizes, the mixer, the hot table, the era draws of layers 4 and 8), and an overlay pays area and clock against hard logic (not quantified here; approximate). The public claim does not change by this layer: it removes one route (per-program bitstreams), it does not lower the chip model's headline row.
## 6. The measurement: compile-ahead per card, and the floor
What a card does between receiving the next seed and swapping (`app/igneum-app/src/engine.rs` `prepare_worker`: export the pack from the node; the worker's `prepare`: generate and compile the program, fill the cache, build the dataset, self-test, optionally race the variants, then swap at the boundary): the per-epoch part is program generation plus kernel compile, the hot table fill (layer 5, per epoch) and the variant race where it runs. The dataset is per day, excluded here; today's worker rebuilds it at every prepare because the pack couples the program to the day, which is a worker change owed before the ladder goes below the base (section 9).
### 6.1 Per card, measured or cited
| Card, compiler | Program (generate + compile) | Hot table fill (layer 5, `docs/plans/hot-table.md`, ca2-cache) | Variant race | Compile-ahead total, race off | Total, race on | Source |
|---|---|---|---|---|---|---|
| Apple M5 Max, Metal | 82 ms (hot-swap entry, 4 October); 58 ms (serve check); 0 to 444 ms over 8 live boundaries (M11 fleet row); 1,798 ms with the Metal compiler cold (variant-racing entry); tonight: 15.9 / 17.7 / 20.4 ms min / median / max over 10 fresh programs, 79 ms for the devnet pack's two libraries (6.2) | 0.07 to 0.22 ms on the GPU | 34.0 / 34.9 / 37.8 s min / median / max over 8 boundaries (M11); 39.8 s one round in the serve check | 0.5 s (1.8 s cold) | 38 s | `docs/bench-log.md`: "first hourly program swap on the live devnet" (4 October), "miner performance: variant racing" (4 October), M11 table (5 October); section 6.2 |
| NVIDIA RTX 5090, NVRTC 12.8 | NVRTC 151 to 180 ms (M11; 151 ms at the PC 2 14:20 boundary); prepare total 580 to 1,074 ms including cache 68, dataset 113 and self-test 511 ms, so about 0.5 to 1.0 s without the dataset; 1,285 ms on the nvcc path of 4 October | 0.67 ms per 256 MiB cache fill on the 5090 scaled to `S / 256`, under 1 ms | 232 to 300 ms to compile 17 variants, 112 s of timing for 3 rounds (M11 race rows, PC 1); one round about 37 s (112 / 3, approximate) | 1.0 s | 38 s | `docs/bench-log.md` M11 table and race rows (5 October), hot-swap entry (4 October), "PC 2 at the 14:20 boundary" (4 October) |
| AMD RX 9070 XT, OpenCL (gfx1201, Adrenalin 26.9.2) | NOT MEASURED at the current worker: `proto-opencl/host.c` times `clBuildProgram` only in the `prepare` path (`buildMs` in the `prepared` line) and no `prepared` line from this card is in any upload (PC 1 app logs 17:26, 18:10, 19:02 UTC; jobs `rdna4-serve-4`, `rdna4-bench-1`, `run-readwidth-9070-20261005c` print cache, dataset and check only) | not measured on AMD | none: the OpenCL worker has no race (`docs/design/miner-tuning.md`: 17 names on NVIDIA, 14 on Metal) | 0.31 s plus the compile (cache 9 ms, self-test 300 ms measured, `rdna4-serve-4`) | the same | OWED (section 9) |
| AMD Radeon integrated gfx1036 (PC 2), OpenCL | prepare total 6.9 / 9.4 / 11.7 s with the 1 GiB dataset build on the iGPU inside; the compile is not separated | | none | under 11.7 s | the same | M11 table; the compile share OWED |
| Intel UHD (US laptop), OpenCL | OpenCL build 3.0 to 6.4 s; prepare total 7.3 / 7.9 / 11.7 s | | none | 6.4 s | the same | M11 table |
| AMD Radeon integrated gfx1036 (PC 1), OpenCL, beside WSL build jobs | prepare total 55.3 / 115.8 / 123.8 s (the dataset build on a loaded iGPU); 2 of 10 boundaries compiled inline | | none | the outlier: 124 s with the dataset | | M11 table |
### 6.2 The Mac, tonight (Measured)
Apple M5 Max (Darwin 25.6.0, 64 GiB), 5 October 2026 21:18 UTC, load average 11 to 14 (other agents' builds and runs; the measure lock held for the 3-s run, `with-lock.sh measure bash scratchpad/epoch-measure.sh`), `proto-metal/igneum-bench` built from this branch (e752fc7 + this document) with `swiftc -O -target arm64-apple-macos11`. Ten distinct programs (seed strings `igneum-devnet-v4-epoch0`, `/epoch1` .. `/epoch9`, version 2 generator, 128 loads per hash), each generated and compiled at run time (`makeLibrary` from source plus `makeComputePipelineState`), dataset 2^28 words, one 2^20 batch and one verify warp per program so the run is the compiles:
./igneum-bench --seed igneum-devnet-v4-epoch0 --hours 10 --dataset-log2 28 --batch-log2 20 --batches 1 --verify-warps 1
| Compile (library + pipeline), ms | Values over the 10 programs |
|---|---|
| each | 18.8, 17.8, 17.6, 16.1, 18.6, 15.9, 17.6, 17.6, 20.4, 18.2 |
| min / median / max | 15.9 / 17.7 / 20.4 |
| library share | 7.7 to 9.8 ms; pipeline 8.2 to 10.6 ms |
| hash rate during the run (GPU time, loaded Mac) | 27.2 to 30.1 MH/s, all 10 verify warps PASS |
The same pack three times through `packbench --pack ../proto-cuda/packs/igneum-devnet-v4-epoch0 --batches 1 --batch-log2 20 --group 256` (two libraries, `memhard.metal` and `program.metal`): compile 79 ms, then 1 ms and 1 ms, the system shader cache answering the identical source; cache fill 0.6 to 0.7 ms GPU, dataset build 20.7 to 20.8 ms GPU for 1 GiB.
Reading: a fresh program costs this card about 18 ms to compile with the Metal compiler service warm, 79 ms for a pack with the dataset kernels, and up to 1.8 s when the compiler is cold (the variant-racing entry's first seed). The fleet's 0 to 444 ms per boundary (M11) sits between those, so the live figure is the app's cold-start states, not the compile itself. None of it is visible against a 600-s window: the Mac's compile-ahead is the race or nothing.
### 6.3 Shares and the floor
The usable window is 600 s on the chain (section 3.1: the VDF output lands 600 s before the epoch on a reference core) and 600 DAA s on the devnet (the stand-in lead; the miner sends `prepare` at about 449 DAA before the boundary, confirm = lead / 4, hot-swap entry).
| Card | Compile-ahead (s) | Share of 600-s epoch | 1,800 | 3,600 | 7,200 | Share of the 600-s window | Fits the devnet's 449-DAA prepare point |
|---|---|---|---|---|---|---|---|
| M5 Max, race off | 0.5 (1.8 cold) | 0.1% (0.3%) | 0.03% | 0.01% | 0.01% | 0.1% | yes |
| M5 Max, race on | 38 | 6.3% | 2.1% | 1.1% | 0.5% | 6.3% | yes |
| RTX 5090, race off | 1.0 | 0.2% | 0.06% | 0.03% | 0.01% | 0.2% | yes |
| RTX 5090, race on (one round) | 38 | 6.3% | 2.1% | 1.1% | 0.5% | 6.3% | yes |
| RX 9070 XT | 0.31 + compile (owed) | | | | | | |
| Intel UHD | 6.4 (11.7 with the dataset) | 1.1% (2.0%) | 0.4% | 0.2% | 0.1% | 1.1% | yes |
| Radeon integrated, PC 2 | under 11.7 with the dataset | 2.0% | 0.7% | 0.3% | 0.2% | 2.0% | yes |
| Radeon integrated, PC 1 under load | 124 with the dataset | 20.7% | 6.9% | 3.4% | 1.7% | 20.7% | yes, 449 s (but it already compiled inline twice at the base) |
The floor by the rule (the slowest card's compile-ahead under 10% of the epoch and inside the window), dataset excluded: 600 DAA s. The slowest measured compile-ahead is the variant race at about 38 s on both the Mac and the 5090, 6.3% of a 600-s epoch and inside the window. Two conditions carry it:
1. The race, if it stays on, costs 6.3% of mining time at the floor against 1.1% at the base. The M11 race rows already found that base wins on both the 5090 and the Mac with the GPU to itself, so the race should default off (or run once a day), which takes the slowest row to 1.0 s. With the race off the floor is set by the Intel iGPU's 6.4-s OpenCL build at 1.1% of 600 s, with the 9070 XT owed.
2. The iGPU tier under load (PC 1's 124 s) fails the 10% rule below 1,240 s because it rebuilds the dataset per prepare. The per-day dataset reuse (section 9) removes that; until it ships, that tier compiles inline at the floor and loses the first minutes of each epoch, as it already did twice at the base.
## 7. Consequences per tier (the standing rule)
| Tier | At the base (3,600) | At the floor (600) | What to do |
|---|---|---|---|
| Home miner, one NVIDIA card, 8 to 32 GB, Windows or Linux | 1 s per hour of prepare, unchanged | 1 s per 10 minutes: 0.2% | nothing; memory unchanged (this parameter adds no bytes) |
| Home miner, one Apple card (M-series), macOS | 0.5 s plus the race's 35 s per hour | race off: 0.5 s per 10 minutes; race on: 6.3% of mining | ship the race default off (M11 finding) |
| Home miner, one AMD card (RDNA 4), Windows or Linux | compile not measured | the same, owed | measure `prepared` on the 9070 XT (section 9) |
| Integrated GPU (AMD or Intel iGPU) | 7 to 12 s per hour, 55 to 124 s under CPU load | 2% of the epoch, or 21% under load with the per-prepare dataset build | per-day dataset reuse in the worker before any signal below the base |
| Rig (several cards, one node; the app) | one pack export per epoch under `EXPORT_LOCK`, cards prepare in parallel | 6x the exports per hour, serialised, each sub-second | nothing |
| Rig on the installer scripts (`packaging/hive/h-run.sh` step 4, `packaging/linux/bin/igneum-miner.sh` step 4, branch `rig-install`) | the miner runs with `--prepare-packs` and `--exit-on-seed-change`: a worker that prepares swaps in place (the hot-swap entry: `rebuilds 0`); exit 42 is the fallback when a prepare misses, and it restarts the miner and re-exports the pack (the loaded iGPU missed 2 of 10 boundaries at the base, M11) | a miss costs a restart plus a pack export plus the inline compile per 10 minutes instead of per hour; a worker with no prepare support (the built-from-source launcher path, `engine.rs` line 2046) restarts at every boundary, six times an hour | the rig installer (a3e7b2b03222f5cff): prepare-ahead must be the only boundary path on the rig before any signal below the base; exit 42 stays as the safety net but a miss at the floor is a 10%-of-epoch loss, so the miss causes (the per-prepare dataset build on a loaded iGPU, the prepare sent late) are fixed first (section 9) |
| Pool user | the pool compiles | the same | nothing |
| Every node's CPU | one core 17% busy on the VDF | one core 100% busy | section 3.3; peers' proofs for a node with no core to spare |
| The chain | difficulty settles 144 s per hour (4%) | 24% of blocks in settle, rate error up to the program step | the ladder's middle steps (1,800: 8%) before the floor; the +-15% settle measurement owed |
## 8. Spec text
### 8.1 Section 1.12, the epoch row and the paragraph after the table (replacing "Epoch `e` covers DAA scores ...")
| Clock | Length | What changes | Label |
|---|---|---|---|
| Epoch | `epoch_len(d)` DAA s, base 3,600; the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200; set by 90% miner signal at a day boundary (section 1.13.1, section 5.7) | The program: new seed words from the VDF of section 4, new kernel | Designed (Counter ASIC 2.0, layer 9, `docs/plans/epoch-length.md`); 3,600 stays the prototype value on every network until a signal moves it |
Epoch `(d, e)` covers DAA scores `[86,400 d + L e, 86,400 d + L (e + 1))` with `L = epoch_len(d)` and `e` in `0 .. 86,400 / L`; every ladder step divides 86,400, so day boundaries are epoch boundaries. The epoch is identified by its start score `s = 86,400 d + L e`. The epoch of a block is the epoch of its own DAA score, and `epoch_len(d)` is a function of the blue blocks of the signalling window that closed at least 2 days before day `d` (section 1.13.1), which are in the header's past; so "which program was this block mined under" is a function of the header alone once the seed is known. The program for epoch `s` is `generate_from_seed_bytes(program_seed_s)` as before, with `program_seed_s` the VDF output of section 4.3 for the checkpoint at least 1,200 DAA s before `s`. `T_epoch` and the 1,200-s lead are genesis constants and do not follow `epoch_len`: the program of every epoch is known 600 s before it starts on the reference core, at every length. At the base `s = 3,600 e` and nothing here differs from the previous text.
### 8.2 Section 1.13.1, one row added to the parameter table, and one paragraph
| Parameter | Base (era 0) | Draw | Bound |
|---|---|---|---|
| Epoch length `epoch_len` | 3,600 DAA s | one draw of the era stream consumed and not used (the value is set by signal) | the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 |
`epoch_len` is the one era-table parameter set by miners rather than by the draw: 90% of blue blocks over a 7-day window carrying the same ladder index (3 bits of the header version) sets that length from the first day boundary at least 2 days after the window closes (section 5.7). It is not a code upgrade: the rule, the ladder and the window are genesis constants, and the chain carries no release. The era stream consumes its draw so that a future draw of this parameter changes no other parameter's value.
### 8.3 Section 4.3, one sentence after step 3
The lead and `T_epoch` are fixed whatever the epoch length of section 1.12: at the floor of 600 DAA s the checkpoint is two epochs back and the program is known one full epoch ahead; at the base it is known for the last sixth of the previous epoch.
## 9. Owed
| Item | What | How |
|---|---|---|
| RX 9070 XT compile time | `clBuildProgram` on gfx1201 at the current worker | a `prepared` line from the card: PC 1 job with `igneum-worker-opencl --serve` on the 9070 XT across one boundary, or `--bench-pack` with a `build` ms print added beside `cache dataset check` in `host.c` (a 2-line change) |
| Radeon integrated compile share | the compile inside the 6.9 to 11.7 s totals | the same print |
| Per-day dataset reuse in the workers | a prepare within the same day keeps the dataset and rebuilds the program only | `proto-metal/main.swift`, `proto-cuda/nvrtc/worker.cpp`, `proto-opencl/host.c`: the pair holds a day key; needed before any signal below the base for the iGPU tier |
| The race default | off, or once a day | the M11 finding; `proto-metal/main.swift` `race = "on"` today |
| The rig scripts' boundary path | prepare-ahead as the only path on a rig; exit 42 kept as the net, never the routine | `packaging/hive/h-run.sh` and `packaging/linux/bin/igneum-miner.sh` (branch `rig-install`, a3e7b2b03222f5cff): a prepare-miss counter in the status line, and the built-from-source launcher path (no prepare support) retired before any signal below the base |
| Difficulty settle at a +-15% step and at 600-s epochs | the share of each epoch in settle at the floor | `sim/difficulty/sim.py` with `EPOCH` as a parameter and the program-step scenario at +-15% |
| Hot table fill on AMD | the per-epoch fill on the 9070 XT | ca2-cache's PC rows |
| The signal's encoding | the 3-bit field in `version` against Kaspa's use of the field | spec 05 section 5.8 |
## 10. Rows for `docs/plans/counter-asic-2.md`
The layer table:
| 9 | Epoch length as a signalled era parameter: base 3,600 DAA s, ladder 600 to 7,200, set by 90% signal at a day boundary, lead and `T_epoch` fixed | a per-program bitstream (an FPGA with a hard datapath): at 600 s nothing it compiles ever runs (42 to 160 min per compile, PRflow FPT 2019; hours on large parts, Aldec) | compile-ahead 1 s per epoch on the 5090, 0.5 s on the M5 Max (38 s with the race on); one CPU core `600 / epoch_len` busy on the VDF | reserve-only tonight: `docs/plans/epoch-length.md` |
The level 3 numbers row:
| Epoch length | 3,600 DAA s at launch; miners can signal it down to 600 (an FPGA defence, no fork) | floor 600: the slowest compile-ahead (the variant race, 38 s) is 6.3% of the epoch and inside the 600-s seed window; at 600 a per-program FPGA bitstream mines 0% of each epoch, at 3,600 up to 47% (42-min compile) | Measured (Mac compile, 5 October), cited (5090, FPGA compile times), owed (9070 XT compile) |

177
docs/plans/era-layout.md Normal file
View file

@ -0,0 +1,177 @@
# Era layout: table layout and working set drawn per era and per program (Counter ASIC 2.0, layers 4 and 8)
5 October 2026, branch `ca2-era`, worker "ca2-era". Status: Designed and Implemented behind the class flag (`LoadClass::era`, not the lottery hash); nothing here changes the default generator, the pinned packs or any live program. Measured sections are marked as such; everything else is design.
Plan: `docs/plans/counter-asic-2.md`, layers 4 ("table layout drawn per era: item size, stride, interleave") and 8 ("working-set size drawn per program"). The era seed is `E_n` of spec 04 section 4.4. Confirmed on 5 October 2026 by grep over `igneum-pow/src` on branches master, readwidth and opencl-rdna4-telemetry: no era draw existed in code before this branch (`igneum-era`, `EraParams`, `era_seed`: no match).
## 1. What is drawn, and from what
Two streams, nothing else:
| Stream | Seeded from | Draws | Sets |
|---|---|---|---|
| Era stream | `seed_words_from_bytes("igneum-era/" \|\| E_n)` words 0 and 1 (spec 01 section 1.13.1 wrote `n_le64 \|\| E_n`; the index is dropped here because `E_n` already commits to `n` through the VDF input of section 4.4 step 2, and the node's seam hands the generator the era bytes alone: `Epoch::from_chain_seeds(epoch, day, era, class, label)`, branch ca2-v3) | 7 per era, fixed | the load width `W`, the stride `(M, R)`, the interleave `pos[0..3]` |
| Program stream | the epoch seed words as today (spec 01 section 1.3.3) | 11 per instruction instead of 9: the 9 of version 2, then 2 window draws (a class whose loads are not version 2's takes the read-width width roll between them, ca2-v3's rule) | per load site: the window shrink `k_off` and its offset `o` |
The era parameters change the class of every program of the era. The program stream does not see the era parameters (two eras with the same epoch seed draw the same instruction list and the same windows, and differ in width, stride and layout); this is what makes the six era packs below a controlled comparison.
### 1.1 Era draw (proposed spec text for section 1.13.1, replacing its parameter table)
One SplitMix64 stream `S` seeded with `lo = words[0] | (words[1] << 32)` of `seed_words_from_bytes("igneum-era/" || E_n)`. Seven draws, in this order, whether or not a value is used:
1. `W = allowed[below(|allowed|)]`: the width in words of every dataset load of the era, drawn from the genesis-fixed ascending set `allowed`, a subset of {1, 4, 16} (4, 16 or 64 bytes). A set of one element pins the width; the draw is still consumed. The set is `{1}` (4 bytes, v2's load): the read-width decision of 5 October 2026 (`docs/plans/read-width.md`, "keep v2; w16 the only width that passes the rules and closes nothing") and the adoption rule of the same evening (a draw that changes the bytes per hash changes the rate; the six-era hash-rate spread must stay under 5 percent per card). The set is recorded in every pack (`IGNEUM_ERA_ALLOWED_WIDTHS`, `program.json` `era.allowed_widths`) and enters the program id. The code keeps the draw general so the set can be widened at genesis without a new derivation.
2. `M = low32(next()) OR 1`: the stride multiplier, odd, so `x -> x * M` is a bijection on 32-bit words.
3. `R = 1 + below(31)`: the stride rotation, in 1..31 (never 0: spec 01 section 1.14 item 3).
4. to 7. `r_i = next()` for `i` in 0..3: the interleave draws. Let `b = log2(W)` (0, 2 or 4) and `free = 4 - b`. Let `c = [b, b + 1, ..., 15]` (16 - b candidates). For `i` in `0..free`: `j = i + (r_i mod (16 - b - i))`, swap `c[i]` and `c[j]`. The interleave is `pos = [0, ..., b - 1] ++ sort(c[0..free])`, four ascending bit positions in 0..15. Draws `r_free..r_3` are consumed and ignored.
The era parameters are `(W, M, R, pos)`. Era 0 of the devnet packs is listed in section 5.
### 1.2 Dataset mapping with the interleave (replaces the last sentence of section 1.8.5)
Word `w` of the dataset holds word `j(w)` of item `t(w)`, where `j(w)` is the 4-bit number whose bit `i` is bit `pos[i]` of `w`, and `t(w)` is `w` with bits `pos[0..3]` removed (the remaining bits in order). With `pos = [0, 1, 2, 3]` this is today's `dataset[w] = item(w >> 4)[w AND 15]`, byte for byte.
Properties kept:
- An item has the same value at every dataset size (item derivation is untouched), and because every `pos[i] < 16`, `dataset[w]` is the same at every dataset size of at least 2^16 words. The 1 GiB vectors of an era remain valid for words below 2^28 at any larger size, as today.
- The low `b = log2(W)` positions are 0..b-1, so the `W` words of one aligned load lie in one item (`t` is the same for all of them and `j` runs `j0 .. j0 + W - 1`). The verifier derives one item per lane per load, as today: the 4,096-item bound of section 1.11 holds (16 loads x 8 iterations x 32 lanes, whatever the width).
- The dataset build writes 16 words of one item to 16 addresses `w(t, j)` (a scatter of 4-byte writes instead of one 64-byte line when `pos != [0, 1, 2, 3]`). This is the only GPU cost of the interleave and is paid once per day; section 6 measures it.
What the interleave does and does not buy. A chip that hard-wires today's layout (64-byte items, a 64-byte line per item) reads the wrong 15 words with every word once the era draws another layout; the layout changes every 180 days inside rules fixed at genesis. A chip whose address decoder can permute 28 address lines under firmware control pays nothing for it. The honest claim is the first sentence only. The stride below is the same kind of lever: two integer operations per load on a chip, nothing on a GPU.
### 1.3 Load address (replaces "a load reads one 4-byte word at `src AND MASK`" in section 1.5 for era programs)
For a load site with window draws `(k_off, o)` (section 1.4), a dataset of `2^D` words (`MASK = 2^D - 1`), and register value `x`:
```
k = min(k_off, D - 26) (0 when D <= 26)
y = rotl(x * M, R)
idx = ((y AND (MASK >> k)) OR ((o AND (2^k - 1)) << (D - k))) AND MASK
base = idx AND NOT (W - 1) (W words from base are folded as verify::fold_words)
```
Uniformity: `x * M` with `M` odd and `rotl` are bijections of the 32-bit word, so `y` is uniform when `x` is; `y AND (MASK >> k)` is uniform on the window; the offset picks which of the `2^k` aligned windows. Branch-free, integer only, three operations before the mask (multiply, rotate, and-or) against one today. The emitted text has one form per dialect, checkable by text search (section 1.14 item 2): CUDA and OpenCL `ds[((rotl_imm(rN * 0x........u, Ru) & 0x........u) | 0x........u) & mask]`, Metal the same with `dataset[` and `& MASK]`.
### 1.4 Window draw per load site (layer 8; proposed text for section 1.4.3)
After the nine draws of version 2 (and the width roll, for a class whose loads are not version 2's), every instruction takes two more draws, used only on a load slot:
```
k_off = below(3) window = the dataset, a half or a quarter of it (2^(D - k_off) words)
o = low32(next()) AND (2^k_off - 1) which aligned window
```
Bounds: the window never goes below `2^26` words (256 MiB; `k = min(k_off, D - 26)` in 1.3), which exceeds the largest on-chip cache of any card in the benchmark (the RTX 5090's 96 MiB L2, the RX 9070 XT's 64 MB Infinity Cache, vendor figures), and never above the dataset. At the prototype dataset (2^28) the windows are 1 GiB, 512 MiB and 256 MiB; at the genesis dataset (2^29) 2 GiB, 1 GiB and 512 MiB. The dataset grows by the step schedule recommended to Josh (spec 01 section 1.13.3 option (b), `docs/analysis/card-lifetime-2026-10-05.md`: power-of-two steps, 4 GiB at year 4, 8 GiB at year 12, 16 GiB at year 28, 32 GiB at year 60, every index `AND MASK`), so the window ceiling follows the steps and the floor stays the genesis constant 2^26 words; nothing in the address of 1.3 needs a range reduction. A program has 16 load sites and so up to 16 windows; the set a program reads is their union (section 7 computes its distribution). The verifier bound is unchanged (1.2).
Why per load site and not per program: a per-program window of a quarter of the dataset would hand a 256 MiB SRAM mirror a third of the hours at the prototype size. Sixteen sites with drawn offsets cover the dataset with high probability (section 7), so the mirror a chip would need is the whole dataset in every hour, and the hour-to-hour variation lands on the memory design (which quarter, which half, how many distinct windows), not on its size.
### 1.5 Program stream (replaces "592 draws per program" in section 1.4.3 for era programs)
16 slot draws, then 64 x 11 = 704: 720 draws per program (768 and 784 for a class with the width roll). On the chain an era program is a class v3 program (branch ca2-v3's seam: `ProgramClass::V3`, generator version 3): its id is `program_id(3, seed words, attempt)` as that branch defines it, and the era it was drawn under is identified beside the id by `IGNEUM_ERA_SEED_HEX` (`packcheck::verify_pack_dir_chain` refuses a pack whose era is not the job's), so the pair (id, era seed) names the program. The experiment classes that are not class v3 (`--class <other> --era ...`) carry the era inside the class id instead: `FNV-1a-64("igneum-program-rw/" || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots || "era/" || allowed[3] || W || M_le32 || R || pos[4])`. The class name is the base name with `-era<first stream word as hex>` (`w4-era401998a5`).
### 1.5.1 The seam (branch ca2-v3, kept as its signatures stand)
`V3_CLASS` is the base class of class v3 (16 loads of 4 bytes, the lottery hash's load; the era rides inside it), `V3_ALLOWED = [1]` its width set. `generate_from_seed_bytes_program_class(label, seed, V3, Some(era))` draws `LoadClass::era(V3_CLASS, era, &V3_ALLOWED)` and stamps generator 3; without era bytes (a template before the era is known) the bare `V3_CLASS` stands. `Epoch::chain_dataset(day, class)` is unchanged: the layout of 1.2 is a property of the program (`program.class.layout()`), applied by the interpreter and by `Epoch::dataset_word`, so the day's cache is shared by every era of a day and the engine keys its caches on `(day, class)` as before.
### 1.6 Acceptance
The rule of section 1.4.6 is unchanged in its tests. Its interpreter mirrors 1.3 at the rule's constant `D = 28` (`idx` as above with `MASK = 0x0fffffff`), as `igneum-pow/src/accept.rs` does. The distinct-address bound counts dataset loads as the read-width branch defines it.
## 2. The era seed on the devnet (stand-in for `E_n`)
Until the 1-hour VDF of section 4.4 is in the node, in the shape of the epoch seed's stand-in (`docs/fork-divergence.md` "Epoch seed"):
| Era | `E_n` |
|---|---|
| 0 | the genesis block hash (32 bytes) |
| n >= 1 | the hash of the last selected-chain block whose DAA score is below `15,552,000 n - 7,200` (the 2-hour lead of section 4.4 step 1) |
The era of a block is `floor(DAA score / 15,552,000)`, a function of the header alone. `E_n` for `n >= 1` is known 7,200 DAA seconds before the era starts, which covers the 1-hour VDF when it arrives and the kernel compile and dataset rebuild now. Test seeds for packs and tests: `E_n` = the 32 bytes (little-endian words) of `seed_words_from_bytes("igneum-era-test/<n>")` (`igneum-pow ... --era igneum-era-test/<n>`, the number only names the pack); raw bytes with `--era <n>:<64 hex>`.
## 3. Memory budget (Josh, 5 October 2026: under 6 GB on an 8 GB card)
The era layout adds no resident memory: a window is a mask and an offset in the kernel text, the interleave is address arithmetic, the stride is two operations. The whole working set on a card, every item from this branch and the others:
| Item | Bytes | Who |
|---|---|---|
| Dataset (prototype) | 1 GiB | existing |
| Cache (256 MiB, resident only while the day's dataset is built, then free; the decided reading, confirmed by the coordinator on 5 October 2026, `docs/analysis/card-lifetime-2026-10-05.md` carries the per-tier working set with the cache freed as the best case) | 256 MiB peak | existing |
| Layer 5 hot table | the ca2-cache worker's figure | not this branch |
| Scratch per resident warp (read-width variant 5, 32 or 128 KiB per warp) | 2,048 warps x 128 KiB = 256 MiB at most | readwidth branch |
| Output and read-back buffers | 2^24 nonces x 8 B = 128 MiB per dispatch | existing harness |
| Era windows, stride, interleave | 0 | this branch |
The era window never exceeds the dataset, so it never grows the footprint.
## 4. Implementation (behind the flag)
| Piece | Where | What |
|---|---|---|
| `EraParams`, `era_draw`, `LoadClass::era` | `igneum-pow/src/generator.rs` | the 7-draw era stream of 1.1; the class carries `era: Option<EraParams>` beside `mix`, `load_slots`, `scratch`, `scratch_kb`; the two window draws per instruction (`Instr::win`, `Instr::off`); the program id of 1.5 |
| `Layout` | `igneum-pow/src/memhard.rs` | `split(w) -> (t, j)`, `join(t, j) -> w`, `LINEAR = [0, 1, 2, 3]`; `MemhardCpu` and `DatasetSource` carry it |
| `load_index` | `igneum-pow/src/verify.rs` | the address of 1.3, shared by the interpreter and the acceptance mirror |
| Emitters | `igneum-pow/src/emit.rs` | the one load form of 1.3 in Metal, CUDA and OpenCL C; `mh_word`, the three `igneum_build` kernels and the Metal build kernel with `mh_t`, `mh_j`, `mh_addr` when the layout is not linear; `IGNEUM_ERA_*` in `program.h`, an `"era"` object in `program.json` |
| CLI | `igneum-pow/src/main.rs` | `--era <igneum-era-test/n \| n:hex>` and `--era-widths 4,16,64` (one width pins) on every command |
| Tests | `igneum-pow/src/*.rs`, `igneum-pow/tests/packs.rs` | the draw is deterministic and within bounds; `split`/`join` are inverse and the dataset is a prefix at every size; six era programs pass the generator contract and the acceptance rule; vectors round-trip; the pinned packs are byte-identical; the six era packs match the emitters and every load has the form of 1.3 |
The default class is untouched: `LoadClass::V2` has `era: None`, every emitter branch on `era` keeps today's text, and `tests/packs.rs` diffs the pinned packs (`igneum-genesis-mh`, `igneum-devnet-v4-epoch0`) against the emitters as before. Section 6 records the diff of a fresh export against the checked-in files.
## 5. The six era packs
`proto-cuda/packs-ca2-era/era-<n>`, `n` in 0..5: the devnet's 32-byte epoch seed `edc4fa84...fb07` and day bytes `igneum-day/20730` (the seeds of the pinned pack `igneum-devnet-v4-epoch0`, which is the v2 baseline with the same program seed), dataset 2^28 words, era seed `igneum-era-test/<n>`, width pinned at 4 bytes (`--era-widths 4`, the default). The program seed is held fixed so that the six packs differ in the era parameters only (section 1); each carries `seeds.txt` for the one-click workers. Every pack is attempt 1 (attempt 0 of this seed is rejected under the era class: 12 draws per instruction give a different stream from v2's). The drawn parameters (`igneum-pow show --epoch-hex edc4... --era igneum-era-test/<n>`, 5 October 2026):
| Pack | Class | Era seed `E_n` (first 16 hex) | W (bytes) | M | R | pos |
|---|---|---|---|---|---|---|
| era-0 | w4-erab2ed8a89 | 5e0587f455a86e91 | 4 | 0x625e5ab3 | 19 | 0, 2, 10, 15 |
| era-1 | w4-era676a17fc | df57136f2ad5f410 | 4 | 0xb2a9d70d | 6 | 1, 3, 8, 13 |
| era-2 | w4-era843155d7 | 7f450623297a954f | 4 | 0x2b4a5b97 | 28 | 1, 3, 4, 8 |
| era-3 | w4-erad6367bfe | 8bffdd3366b9c3ff | 4 | 0x27ea7eff | 30 | 2, 3, 8, 13 |
| era-4 | w4-era4488f3ed | e593fc1d48475c88 | 4 | 0x4d38603d | 10 | 2, 9, 13, 15 |
| era-5 | w4-eraf897c84e | ff87ad96a1b53f36 | 4 | 0x03ac37ad | 22 | 0, 2, 10, 13 |
All six are class v3 packs (generator 3, `IGNEUM_PROGRAM_CLASS "v3"`, `IGNEUM_ERA_SEED_HEX`), attempt 0, program id `73bcbfe8ccf988f1` in every pack (the seam's `program_id(3, seed, attempt)`; the era seed beside it names the program), the era layout over version 2's item construction (mixer x1, the genesis cache) so that the v2 baseline pack is the same dataset; the chain's class v3 composes the same draw over `LoadClass::MX4` (mixer x4, growth), and the integration re-exports these packs on it after the PC rows. The era-seed-to-pack assignment above is from `program.h` of each pack; the test-seed numbering is only the pack name.
The windows are a property of the program, so they are the same in all six packs (site:shrink:offset): `1:2:3 4:2:2 6:0:0 10:1:0 12:1:1 20:2:0 27:1:0 30:0:0 35:0:0 40:0:0 41:0:0 43:2:3 45:1:0 52:0:0 53:2:2 54:0:0`: 7 sites read the whole dataset, 5 a half, 4 a quarter; the union is the whole dataset.
## 6. Measurements
Pending at the time of this commit; each table below says the machine, the date, the harness and the command when filled.
### 6.1 Byte-identical default path
### 6.2 Bit-exactness on the Mac (Metal, Apple OpenCL, CUDA emulation)
### 6.3 Hash rate per era on the M5 Max (Metal) and the CPU verifier
### 6.4 PCs (RTX 5090 CUDA, RX 9070 XT OpenCL): prepared, waiting for the go
## 7. The chip-model line per draw
From `docs/analysis/m16-recompute-attacker-2026-10-05.md` and the random-read ceilings of `docs/bench-log.md` ("the 9070 XT on the eGPU": 9070 XT 2.42 to 2.68 G loads/s at 1 GiB, 5090 16.4 to 18.0, M5 Max 3.41 to 3.49; every random 4-byte read costs AMD a 64-byte line):
| Quantity | Formula |
|---|---|
| Bytes read per hash | 128 loads x W bytes |
| Distinct 64-byte lines per hash | 128 (one line per load at every W up to 64 bytes; the census's 120 to 128 distinct addresses per hash) |
| SRAM a chip needs to mirror what the hash reads | the union of the program's 16 windows (a distribution over programs; section 7.1) |
| Latency-bound share | measured rate / (the card's 4-byte random-read ceiling / 128) |
The latency-bound share uses the 4-byte ceiling for every width because the 9070 XT line probe showed the same count per second for 4-byte and 64-byte random reads; the 5090's 16-byte and 64-byte ceilings are the read-width branch's measurement, cited when they land.
### 7.1 Union of windows per program
Filled from a CPU census over programs (section 6).
## 7.2 Found on the way (harness defects, both fixed on this branch)
| Where | Defect | Fix | Checked |
|---|---|---|---|
| `proto-cuda/nvrtc/packfile.h` (the one-click workers) | the seed words were re-derived from the bare epoch seed, so every pack of attempt 1 or higher was refused ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT"); 5.14 percent of epochs under v2, all six era packs, and the epoch 34 fleet outage of 18:23Z on 5 October 2026 (branch pack-loop af983a7, which this branch takes: `pf_program_words`) | the pack-loop derivation merged over the readwidth packfile (class fields and string seeds kept) | the devnet pack (attempt 0) loads, the six era packs (attempt 1) load, a copy of era-0 with the attempt tampered to 0 is refused on the re-derivation (section 6) |
| `proto-cuda/host.cu`, `proto-opencl/host.c` | the host-side dataset word was `mh_item(w >> 4)[w AND 15]`, the harness's own copy of the linear layout; under an interleaved layout the "64 random points vs host derivation" check failed while the Mac samples and the vectors passed | `host_ds_word` calls the pack's `mh_word` (memhard.h), which carries the layout | `proto-cuda/emu/test-layout.sh`: the CUDA emulation on era-1 (interleaved) and the devnet pack (linear) must pass the random-point check; it failed on era-1, era-3 and era-5 before the fix (section 6) |
## 8. What is unverified
- Everything in section 6 marked pending.
- The 1-hour VDF does not exist; the devnet stand-in of section 2 is a proposal.
- The interleave's value against a chip with a programmable address decoder is nil (1.2); the claim is limited to hard-wired layouts.
- The window floor of 2^26 words is set by the 5090's L2 (96 MiB) and the 9070 XT's Infinity Cache (64 MB, vendor figures); a future card with a larger cache moves the floor, which is a genesis constant.
- No cryptanalysis of the stride (a multiply and a rotate before the mask); it is a bijection, so the address distribution is that of the register value, as today.

266
docs/plans/hot-table.md Normal file
View file

@ -0,0 +1,266 @@
# Hot table: a second table sized to GPU cache, read beside the 1 GiB dataset
Counter ASIC 2.0, layer 5 (`docs/plans/counter-asic-2.md`). Experiment branch `ca2-cache`, 5 October 2026 (night), on top of the read-width branch (`readwidth` b970dda: `LoadClass`, `verify::fold_words`, the scratch variant, `proto-metal/packbench.swift`, `proto-opencl/host.c --bench-pack`). Nothing here is the lottery hash: every hot class sits behind the generator flag and the default v2 path is byte-identical (the pinned packs `igneum-genesis-mh` and `igneum-devnet-v4-epoch0` are diffed by `igneum-pow/tests/packs.rs`).
## 1. The idea
The honest hash is 128 dependent random 4-byte reads over 1 GiB (spec 01 section 1.5). A chip that wants a gain on that must beat a GPU at DRAM random reads, which is the same physics for both (the plan's last section). What a chip can do that a GPU cannot is choose its memory: a chip can mirror read-only data into SRAM and serve it at SRAM latency, if the data fits.
The hot table turns that around. A second table `H` of `S` MiB (32, 64 or 96 in this experiment) is derived from the epoch seed and read by `k` of the 16 load slots (2, 4 or 8). `S` is chosen to fit the caches of the cards that mine: 96 MiB L2 on the RTX 5090, 64 MB Infinity Cache plus 8 MB L2 on the RX 9070 XT, the system level cache on the M5 Max (figures approximate, from memory; the probe of section 6 measures what each card does at each size). A GPU gets the hot loads as cache hits for free. A chip must carry `S` MiB of SRAM for the same hits, beside the DRAM path it still needs for the other `16 - k` slots, or serve `H` from DRAM and fall behind the GPU by the hot share. Either way the chip pays for something the GPU already has.
What this does not change: the cold loads (the 1 GiB dataset) stay a dependent chain at DRAM latency, so the latency-bound property holds for them; the hot loads are interleaved in the same chain (every load's address is a fresh register of the same iteration, G2), so a hot hit shortens the chain by one DRAM latency and nothing else.
## 2. The specification text (proposed; prototype values, to be fixed at gate 1)
### 2.1 Hot key and fill
For an epoch whose program seed bytes are `e` (spec 01 section 1.12: the UTF-8 of a seed string in the packs, the 32-byte VDF output on the chain; the bytes before the attempt suffix of 1.4.6, so every attempt of one epoch shares one table):
```
KH = seed_words_from_bytes("igneum-hot/" || e) 8 words
```
`H` has `N_H = S x 2^18` words (`S` MiB) in `N_seg = S x 256` segments of 64 chained lines of 16 words, filled exactly as the cache of section 1.8.3 with `KH` in place of `K` and the tag `("Igne", "umHT") = (0x49676e65, 0x756d4854)` in place of `("Igne", "umMH")`:
```
prev = 0^16
for j in 0..63:
in = prev XOR (sigma[0..3] || KH[0..7] || s || j || tag[0..1])
line = B(in) ChaCha12 with feed-forward, section 1.8.2
H[segment s, line j] = line
prev = line
```
One GPU thread per segment, as the cache fill. The fill is a once-per-epoch cost: `S / 256` of the 256 MiB cache fill (section 1.8.3 table: 0.6 to 2.1 ms on the M5 Max GPU, 0.67 ms on the RTX 5090, 175 to 190 ms on one CPU core for 256 MiB), so under 1 ms on a GPU and 22 to 71 ms on one core, measured below.
### 2.2 The hot load
A hot load slot reads `H` instead of the dataset with the same fold (width 1: a plain XOR):
```
hot: dst = dst XOR H[mulhi(src, N_H)]
```
`mulhi(a, b)` is the high 32 bits of the 64-bit product (the `mulhi` family of section 1.4.1, bit-exact on Metal, CUDA and OpenCL). The index lies in `[0, N_H)` for any `N_H`, which is what lets `S = 96` exist: 96 MiB is not a power of two, so `src AND MASK` cannot address it. This is the multiply-shift range reduction section 1.13.3 proposes for the growing dataset, so the hot table is also its first measured use. For `S = 32` and `64` the mapping takes the top 23 or 24 bits of `src` where the dataset load takes the low 28; a fresh source is uniform, so neither choice costs uniformity (section 6, hot-load uniformity test).
The emitted text has one form per dialect, checkable by text search as the mask check of section 1.14 item 2: `hot[mulhi(rN, HOT_WORDS)]` (Metal), `hot[__umulhi(rN, HOT_WORDS)]` (CUDA), `hot[mul_hi(rN, HOT_WORDS)]` (OpenCL), with `HOT_WORDS` a literal of the pack. A hot pack's hash kernels carry exactly `16 - k` masked dataset loads and exactly `k` hot loads (`igneum-pow/tests/packs.rs`).
### 2.3 Which slots are hot
The 16 load slots are drawn first by the partial Fisher-Yates of section 1.4.3, which emits them in a uniformly random order. The first `k` slots in that draw order are the hot slots. Drawn, not fixed, because a fixed pattern (every fourth load, say) would let a pipeline schedule its SRAM reads statically for every hour; drawn costs no extra draw, so a `hot(S, k)` program takes the version 2 stream exactly and is the version 2 program with `k` of its loads redirected (the width roll of the read-width classes is not taken: `LoadClass::takes_width_roll`). A hot slot keeps every rule of a load: the fresh-source draw (G2), the injecting-write count (acceptance (b)), the lane-constant test and the distinct-address count (acceptance (c)); its address is tagged apart from dataset addresses in the count so a hot word and a dataset word at the same index are two addresses.
The class composes with the read-width fields of `LoadClass` (width mix, scratch `k` and `kb`): the hot slots are taken from the drawn slots after the scratch slots, so a class may carry a width mix, a scratch and a hot table at once. The experiment packs below use the version 2 widths and no scratch.
Two forms (coordinator's decision, 5 October 2026, after the first Mac measurement):
| Form | Load slots drawn | Dataset loads per hash | Hot loads per hash | Items per warp (verifier bound) | Class name | What a chip pays |
|---|---|---|---|---|---|---|
| replaced (`hot(S, k)`) | 16, the first `k` in draw order hot | `(16 - k) x 8` | `8k` | `(16 - k) x 256` | `hot64k4` | the on-die-cache recompute chip of `docs/analysis/scratch-soundness.md` 3.4 (the 256 MiB cache in SRAM, items derived on the fly, 333 MH/s at 50 T op/s, 2.4x the 5090) GAINS: a hot load replaces an item derivation (1,170 ops) with an SRAM read, so at `k = 4` its rate rises 1.33x against the GPU's measured 1.05 to 1.22x |
| added (`hot(S, k, added)`) | `16 + k`, the first `k` in draw order hot | `16 x 8 = 128` | `8k` | 4,096, unchanged | `hot64k4a` | `S` MiB of SRAM and `k` reads per iteration for nothing: the 16 item derivations stay; the GPU pays `k` cache hits |
The added form is the one that taxes the named chip; the replaced form stays as the measured record (section 6). In the added form the slot draw is a partial Fisher-Yates of `16 + k` slots over 1..63, so the program stream differs from version 2 (another slot count), and the width roll is still not taken (`takes_width_roll`: the slot count is 16 plus the class's own added hot slots). `loads_per_hash` is `128 + 8k`; the acceptance rule's distinct-address bound scales with it as for the read-width classes.
Program id: `FNV-1a 64 over "igneum-program-rw/" || generator || seed words || attempt || mix || load_slots || "hot/" || S || k [|| "added"]` (the read-width id with the hot fields appended), so no hot pack can be mistaken for a version 2 one, for another hot class or for the other form.
### 2.4 Acceptance (section 1.4.6)
The dynamic test stays a pure function of the program. A hot load reads `dataset_elem(mulhi(src, N_H), SW[2], SW[3])` (the six-operation closed form keyed by seed words 2 and 3, where the dataset stand-in is keyed by words 0 and 1), so the two stand-ins are two tables without any cache or day. Every other test is unchanged; the distinct-address bound is the read-width rule's (dataset and hot loads counted, scratch read-modify-writes not).
### 2.5 The verifier (section 1.11)
A verifier holds, per day, the mixer parameters and the 256 MiB cache, and, per epoch, `H` (`S` MiB, filled on one core in the times of section 6). A hot load is one table read per lane; a dataset load is the item derivation of section 1.11 as before. The verifier never holds the dataset. Per epoch the verifier's memory is `256 MiB + S MiB`.
## 3. Memory budget (Josh's cap, 5 October 2026: the working set on a card stays under 6 GB on an 8 GB card)
Decided by the coordinator on 5 October 2026 after the card-lifetime review (`docs/analysis/card-lifetime-2026-10-05.md`, branch `card-lifetime` 1fecfe2, merged into `ca2-coord`): the GPU frees the 256 MiB cache after the daily dataset build (the hash never reads it; the rebuild costs 0.67 ms of fill and 13.4 ms of build on the 5090, bench-log 3 October 2026), so the cache is not resident and the working set is dataset + hot table + scratch (0 in v3) + buffers. The table below is that review's per-tier table with the hot table at its largest (96 MiB), buffers 128 MiB (the harnesses' 2^24-nonce output), resident warps = SMs x 48 on NVIDIA Ampere and later (SM counts approximate, from memory; the GTX 1650 is Turing at 32 warps per SM), 2,048 launched warps on Apple (the Metal harness). The scratch columns are the read-width variant's 32 and 128 KiB per resident warp, kept for the record; v3 carries no scratch.
| Tier | Card assumed (SMs, approximate) | Resident warps | Scratch at 32 KiB | Scratch at 128 KiB | Hot table | Dataset at genesis | Buffers | Total, no scratch | Total at 32 KiB | Total at 128 KiB | Years the dataset leaves under mapping (b), cache freed |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 4 GB | GTX 1650 (14 SMs x 32) | 448 | 14 MiB | 56 MiB | 96 MiB | 2,048 MiB | 128 MiB | 2,272 MiB | 2,286 MiB | 2,328 MiB | to year 4 (the 4 GiB step) |
| 8 GB | RTX 3050 (20) | 960 | 30 | 120 | 96 | 2,048 | 128 | 2,272 | 2,302 | 2,392 | to year 12 (the 8 GiB step) |
| 12 GB | RTX 3060 (28) | 1,344 | 42 | 168 | 96 | 2,048 | 128 | 2,272 | 2,314 | 2,440 | to year 28 (the 16 GiB step; year 12 if the cache were resident) |
| 16 GB | RTX 5060 Ti (36) | 1,728 | 54 | 216 | 96 | 2,048 | 128 | 2,272 | 2,326 | 2,488 | to year 28 |
| 24 GB | RTX 4090 (128) | 6,144 | 192 | 768 | 96 | 2,048 | 128 | 2,272 | 2,464 | 3,040 | to year 60 (the 32 GiB step) |
| 32 GB | RTX 5090 (170) | 8,160 | 255 | 1,020 | 96 | 2,048 | 128 | 2,272 | 2,527 | 3,292 | to year 60 |
| Apple 8 to 64 GB | M-series, 2,048 launched | 2,048 | 64 | 256 | 96 | 2,048 | 128 | 2,272 | 2,336 | 2,528 | 8 GB to year 4, 16 GB to year 12, 32 GB to year 28, 64 GB to year 60 (50% of unified memory usable, the review's assumption) |
The prototype packs here use the 1 GiB dataset of spec 1.5 (1,024 MiB less in every total). Every total is under 6 GB with the hot table at its largest, so `S` is not what the cap binds: the dataset's growth is, and the hot table takes 96 MiB of the room at every tier (about 2 months of the 0.5 GiB-a-year schedule). The verifier's memory is `256 MiB + S MiB` (section 2.5); the GPU's is the table above.
## 4. The chip model with the SRAM it would need
Figures: SRAM area per bit from `docs/analysis/m16-recompute-attacker-2026-10-05.md` section 3 (256 MiB in about 100 to 300 mm^2 at a current node: the low end from a 0.02 um^2 bit cell with array overhead, the high end from wafer-scale parts at about 1 MB per mm^2; approximate, from memory) and from `docs/plans/counter-asic-2.md` (256 MB in about 45 mm^2 at a leading node, approximate). That is 0.18, 0.39 and 1.17 mm^2 per MiB. Die areas: a 750 mm^2 GPU-class die (the M16 model's equal-silicon comparison) and a 100 mm^2 memory-chip die (an assumption for a latency-bound chip whose die holds memory controllers and little else; labelled as such).
| S (MiB) | SRAM at 0.18 mm^2/MiB | at 0.39 | at 1.17 | Share of a 100 mm^2 die (0.39) | Share of a 750 mm^2 die (0.39) |
|---|---|---|---|---|---|
| 32 | 6 mm^2 | 12 mm^2 | 37 mm^2 | 11% | 1.6% |
| 64 | 12 mm^2 | 25 mm^2 | 75 mm^2 | 20% | 3.2% |
| 96 | 17 mm^2 | 37 mm^2 | 112 mm^2 | 27% | 4.8% |
The gain arithmetic. Let `G0` be a chip's gain over a GPU on the version 2 hash (the plan's public claim: under 2x; the plan's last section says the bound is DRAM latency, the same physics on both). Let `g` be the GPU's own measured speed-up from the hot class over version 2 on the same card (section 6: `hash rate hot(S, k) / hash rate v2`). The ideal `g` with every hot load a hit at zero cost is `16 / (16 - k)`: 1.14x at `k = 2`, 1.33x at `k = 4`, 2.0x at `k = 8`; the measured `g` says what share of that a real cache delivers while the 1 GiB dataset streams through the same cache.
| Chip | Hot loads served from | Gain after the hot table | Arithmetic |
|---|---|---|---|
| A, no SRAM for H | DRAM | `G0 / g` | the chip's rate is what it was; the honest GPU gained `g` |
| B, S MiB of SRAM for H | SRAM | `G0 x A_die / (A_die + A_S)` per unit of silicon | the chip regains `g` and pays `A_S` on top of its die |
| B on a 100 mm^2 die, S = 64, 0.39 mm^2/MiB | SRAM | `0.80 x G0` | 100 / 125 |
| B on a 750 mm^2 die, S = 96, 0.39 mm^2/MiB | SRAM | `0.95 x G0` | 750 / 787 |
What the big-table loads still cost the chip: `(16 - k) x 8` dependent DRAM reads per hash at the card's loaded latency (the 9070 XT entry's probe: 451 ns on the 5090 at 4,096 lanes, 276 ns unloaded on the 9070 XT, 1,949 ns on the M5 Max at 4,096 lanes; section 6 repeats the probe here). At `k = 4` that is 96 reads per hash; to match one RTX 5090 at its honest 229 Mhash/s (the M16 model's reference) a chip must keep `229 M x 96 x 300 ns = about 6,600` DRAM reads in flight at a 300 ns latency, and 11,000 at 500 ns, whatever its arithmetic. That queue depth is a memory-controller property, which is the latency-bound argument of the plan restated for the cold share.
What the hot table does to the recompute attacker of M16: nothing good. `H` is a plain ChaCha12 chain, so a word of it can be recomputed from `KH` at `j + 1` block evaluations (32.5 on average, about 40,000 integer operations per hot load, approximate, against 1,170 per dataset item), which is 35x the cost of recomputing a dataset word. A chip stores `H` or reads it from DRAM; it does not recompute it.
Reading, before the measurements: the hot table costs a chip `A_S` of die area or `g` of rate. Which one binds depends on the measured `g`, which is why the measurement comes first. If a GPU's cache delivers most of the ideal `g` with the dataset streaming beside it, `k = 8` at `S = 64` doubles the honest rate and halves chip A. If the cache delivers little, the hot table is a cost to the verifier (`S` MiB per epoch) with no gain, and the layer is dropped.
## 5. Implementation (branch `ca2-cache`)
| Where | What |
|---|---|
| `igneum-pow/src/generator.rs` | `LoadClass { hot: Option<HotClass> }` beside `mix`, `load_slots`, `scratch`, `scratch_kb`; `HotClass { mb, k }`; names `hot32k4`, `hot64k2`; `Op::Hot`; the first `k` drawn load slots after the scratch slots are hot; no width roll for a hot class with version 2 widths (`takes_width_roll`); the program id carries `hot/S/k` |
| `igneum-pow/src/memhard.rs` | `HotTable` (key, S, words), `hot_key(seed_bytes)`, `hot_words(mb)`, `hot_segments(mb)`, `hot_index(src, words)`, the tagged segment fill shared with the cache, `HOT_TAG` |
| `igneum-pow/src/verify.rs` | `DatasetSource::hot: Option<HotTable>`; `Op::Hot` in `step`; `Epoch::new_class` and `from_seed_bytes_class` fill `H` from the program's seed bytes when the class has a hot table |
| `igneum-pow/src/accept.rs` | `Op::Hot` with the closed-form stand-in of 2.4 |
| `igneum-pow/src/emit.rs` | the hot load statement in the three dialects; `hot` as the buffer after the init words (Metal buffer 3, or 4 when bound; CUDA and OpenCL argument after `mask`, or after the init words when bound) and before the scratch triple; `HOT_WORDS` literal; `ht_cache_segment` and `igneum_hot_fill` kernels in memhard.h, kernel.cu, kernel.cl and memhard.metal; `program.h` `IGNEUM_HOT_MB`, `IGNEUM_HOT_WORDS`, `IGNEUM_HOT_SEGMENTS`, `IGNEUM_HOT_SLOTS`, `IGNEUM_HOT_KEY_INIT`; `vectors.h` and `vectors.json` the hot head, last line and FNV-1a 64 |
| `igneum-pow/src/main.rs` | `--class hot<S>k<k>` on every command (`bench` reports the fill time and the verifier ms per warp) |
| `igneum-pow/tests/packs.rs` | the five hot packs pinned (program, vectors, every emitted file, the load-form count: `16 - k` masked loads and `k` hot loads) |
| `proto-cuda/packs-ca2-hot/` | replaced: `hot32k4`, `hot64k4`, `hot96k4`, `hot64k2`, `hot64k8`; added: `hot32k4a`, `hot64k4a`, `hot96k4a`; all from seed `igneum-genesis`, day `2026-10-03` |
| `proto-cuda/nvrtc/packfile.h` | `hotMb`, `hotWords`, `hotSegments`, `hotSlots`; the hot self-test values; `pf_selftest` checks them when the pack carries them |
| `proto-opencl/host.c` | `--bench-pack` and the self-test allocate and fill `H` from the pack's `igneum_hot_fill`, check its head, last line and FNV-1a 64, pass it as the argument after the init words |
| `proto-metal/packbench.swift` | the same on Metal from `memhard.metal` |
| `proto-cuda/nvrtc/worker.cpp` | `--check` and `--bench` (new: the base kernel timed over `--batches` dispatches, one RESULT line) allocate, fill and self-test `H`; the launch passes it |
## 6. Measurements
Every row names the machine, the harness and the command; the bench-log entry of 5 October 2026 ("the hot table on the M5 Max") carries the raw lines. The Mac rows were taken on 5 October 2026, 20:19 to 20:21 UTC, under the measure lock, with the Mac's load average at 14 to 27 from other agents' CPU work (the lock serialises builds and measurements, not every process), so they are ordered, repeatable to within a few percent against each other, and not the Mac's quiet numbers. The PC rows wait for the coordinator's go.
### 6.1 Random-read probe at the hot sizes
`igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S` (`proto-opencl/host.c`; dependent random 4-byte loads, best of 3, 256 steps per lane, work-group 256; the ceiling is the 4,194,304-lane row; Apple OpenCL reports wall time).
| Card | 32 MiB ceiling | 64 MiB | 96 MiB | 1024 MiB | Ratio 32 / 1024 | 64 / 1024 | 96 / 1024 | ns per dependent load at 4,096 lanes (32, 64, 96, 1024 MiB) |
|---|---|---|---|---|---|---|---|---|
| M5 Max (Apple OpenCL) | 21.7 G loads/s | 12.8 | 12.3 | 3.50 | 6.2 | 3.7 | 3.5 | 1,168; 1,129; 1,242; 1,844 |
| RTX 5090 (CUDA worker `--memprobe`, PC 1, job run-ca2-hot-5090-20261005, card off in the app) | 112.6 | 112.6 | 112.6 | 17.6 | 6.4 | 6.4 | 6.4 | 320; 340; 336; 610 |
| RX 9070 XT (OpenCL worker `--memprobe --device 1`, PC 1, job run-ca2-hot-9070-20261005, card off in the app) | 9.88 (10.8 at 262,144 lanes) | 9.47 | 8.18 | 2.43 | 4.1 | 3.9 | 3.4 | 396; 457; 454; 1,579 |
Reading, RTX 5090: all three sizes sit inside the 96 MiB L2 at one ceiling (112.6 G loads/s, 6.4x the DRAM figure), so the probe alone promises a full hit rate for every S. RX 9070 XT: 32 and 64 MiB inside the Infinity Cache at 9.5 to 10.8 G loads/s (3.9 to 4.1x), 96 MiB at 8.2 (3.4x), as the 9070 XT entry's 64 MiB row said. Streams: 5090 1,563 GB/s at 1024 MiB, 9070 XT 633 GB/s (both at their rated figures, the cards were not parked).
Reading, M5 Max: the step from 32 to 64 MiB halves the ceiling (21.7 to 12.8 G loads/s) and 96 MiB sits with 64, so 32 MiB is inside a cache level that 64 MiB is not (the M5 Max's system level cache size is not published; approximate reading: the 32 MiB table fits, the two larger ones mostly do not and run at a 3.5 to 3.7x advantage over DRAM from whatever hits they get). The coalesced stream grows with the buffer (139, 199, 243, 522 GB/s) because the small buffers are read once from cold.
Predicted `g` from the probe alone, with a hot load costing `1 / ceiling_S` and a cold load `1 / ceiling_1024`: `g = 1 / ((16 - k) / 16 + (k / 16) x ceiling_1024 / ceiling_S)`.
### 6.2 Bit-exactness and hash rate per hot pack
Metal: `proto-metal/packbench --pack <dir> --batches 5 --batch-log2 24` (GPU time). Apple OpenCL: `igneum-bench-cl-igneum-genesis-mh --bench-pack --pack <dir> --batches 5 --batch-log2 24` (wall time). Both harnesses fill the hot table on the device from the pack's `igneum_hot_fill` and check its head, last line and FNV-1a 64 against vectors.json or vectors.h; vectors are the 96 lanes of the Rust reference; the fingerprint is FNV-1a 64 over the 2^24 outputs at base nonce 0.
| Pack | Vectors (Metal, OpenCL) | Hot table FNV (Metal, OpenCL) | Fingerprint 2^24 (both harnesses equal) | Metal Mhash/s | Apple OpenCL Mhash/s | g against v2 (Metal) | Predicted g from the probe | Ideal g |
|---|---|---|---|---|---|---|---|---|
| v2 (igneum-genesis-mh) | 3/3, 96/96 | none | 25f96e7dce90bd4e | 27.68 | 27.61 | 1 | 1 | 1 |
| hot32k4 | 3/3, 96/96 | PASS, PASS | d2e6cf3b61d0b9fe | 33.90 | 33.92 | 1.22 | 1.27 | 1.33 |
| hot64k4 | 3/3, 96/96 | PASS, PASS | e4c5263ac650cc0d | 30.93 | 30.89 | 1.12 | 1.22 | 1.33 |
| hot96k4 | 3/3, 96/96 | PASS, PASS | 5d63439b6e394521 | 29.06 | 28.97 | 1.05 | 1.22 | 1.33 |
| hot64k2 | 3/3, 96/96 | PASS, PASS | 352633bdbbb0d2b6 | 27.67 | 27.27 | 1.00 | 1.10 | 1.14 |
| hot64k8 | 3/3, 96/96 | PASS, PASS | da54630d7dfaaf85 | 47.42 | 46.73 | 1.71 | 1.57 | 2.0 |
| hot32k4a (added) | 3/3, 96/96 | PASS, PASS | 8a3414735db4523c | 25.76 | 25.72 | 0.93 | 0.96 | 1 |
| hot64k4a (added) | 3/3, 96/96 | PASS, PASS | 45668f34105f6307 | 23.92 | 23.87 | 0.87 | 0.94 | 1 |
| hot96k4a (added) | 3/3, 96/96 | PASS, PASS | af763997dfee4c82 | 22.92 | 22.88 | 0.83 | 0.93 | 1 |
For the added form the ideal `g` is 1 (the 16 dataset loads stay) and the probe predicts `g = 1 / (1 + (k / 16) x ceiling_1024 / ceiling_S)`: 0.96 at 32 MiB, 0.94 at 64 and 96 MiB for `k = 4` on the M5 Max; what matters is how far below 1 the GPU lands (its cost of the layer) against the chip's `S` MiB of SRAM and `k` reads.
The PCs (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7 on app 0.3.9 before and after, the card under test switched off in the app through `api/cards` and restored; CUDA worker `--bench --batches 5 --batch-log2 24 --block-warps 1` on the RTX 5090, wall time around the stream sync; OpenCL worker `--bench-pack --batches 5 --batch-log2 24 --device 1` on the RX 9070 XT, device event time, work-group 256; the v2 references are the readwidth entry's same-night, same-worker numbers: 136.1 and 18.15 MH/s). Every pack bit-exact with the Mac's fingerprint, hot table head, last line and FNV PASS, 96/96 lanes, on both cards.
| Pack | RTX 5090 MH/s | g (v2 136.1) | probe-predicted g (5090) | RX 9070 XT MH/s | g (v2 18.15) | probe-predicted g (9070 XT) | ideal g |
|---|---|---|---|---|---|---|---|
| hot32k4 | 146.6 | 1.08 | 1.27 | 19.79 | 1.09 | 1.23 | 1.33 |
| hot64k4 | 140.8 | 1.03 | 1.27 | 18.73 | 1.03 | 1.23 | 1.33 |
| hot96k4 | 138.5 | 1.02 | 1.27 | 18.33 | 1.01 | 1.21 | 1.33 |
| hot64k2 | 137.5 | 1.01 | 1.13 | 18.17 | 1.00 | 1.11 | 1.14 |
| hot64k8 | 163.6 | 1.20 | 1.73 | 22.32 | 1.23 | 1.59 | 2.0 |
| hot32k4a (added) | 118.7 | 0.87 | 0.96 | 15.27 | 0.84 | 0.94 | 1 |
| hot64k4a (added) | 115.4 | 0.85 | 0.96 | 14.62 | 0.81 | 0.94 | 1 |
| hot96k4a (added) | 114.4 | 0.84 | 0.96 | 14.56 | 0.80 | 0.93 | 1 |
The 5090 at `--block-warps 8` (6 blocks per SM, 8,160 resident warps) is within 0.7% of every row above; the 9070 XT at work-group 32 within 0.5%. Hot fill on the 5090: 1.2 to 2.5 ms (wall, driver API); on the 9070 XT 1.6 to 5.3 ms.
Reading, the PCs. The probe promised a full hit rate for every S on the 5090 (all three tables inside the 96 MiB L2 at one ceiling) and 3.4 to 4.1x on the 9070 XT, and the hash delivered a fraction of it: 1.02 to 1.08x at `k = 4` against the probe's 1.27x and the ideal 1.33x on the 5090, 1.01 to 1.09x on the 9070 XT, 1.20 to 1.23x at `k = 8` against 1.73 and 2.0x. The table that stands alone in the probe does not stand up with the 1 GiB dataset streaming through the same cache: the dataset's random lines evict it (the 5090's L2 and the 9070 XT's Infinity Cache are shared by every load; neither card partitions them). The added form costs the 5090 13 to 16% and the 9070 XT 16 to 20% of its rate for four extra loads per iteration, three to five times the probe's 4 to 7%. The three cards agree on the shape; the 5090's larger cache buys it nothing over the Mac at 32 MiB (1.08 against 1.22).
Hot table fill on the M5 Max GPU (Metal, GPU time): 0.07 ms at 32 MiB, 0.15 ms at 64 MiB, 0.22 ms at 96 MiB.
Reading, M5 Max. Bit-exactness holds: Metal and Apple OpenCL give one fingerprint per pack and every vector lane against the Rust reference, hot table included. On the rate, the hot loads are worth 92% of the probe's prediction at 32 MiB (1.22 against 1.27), 50% at 64 MiB (1.12 against 1.22 is 0.12 of 0.22) and 23% at 96 MiB; at `k = 8` the measured 1.71 is above the prediction (1.57), which says the cold loads also go faster when half of them are gone (fewer dependent DRAM reads in the chain per hash). `hot64k2` gained nothing on this Mac at load (27.67 against 27.68). So on Apple silicon, with the 1 GiB dataset streaming through the same cache, the table that fits (32 MiB) delivers most of its ideal and the larger ones lose most of theirs. The 5090 and 9070 XT, whose caches are the plan's targets, are the PC job.
### 6.3 CPU verifier, one core, avg of 50 warps
`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class <class> --warps 50` (release build, one M5 Max core, the Mac at load as above; the v2 row from the same session, the 0.604 ms of the brief was an earlier quiet run).
| Class | Hot fill, one core | Items derived per warp | ms per warp | Against v2 |
|---|---|---|---|---|
| v2 | none | 4,096 | 0.626 | 1 |
| hot32k4 | 24.0 ms | 3,072 | 0.489 | 0.78 |
| hot64k4 | 46.4 ms | 3,072 | 0.504 | 0.81 |
| hot96k4 | 73.0 ms | 3,072 | 0.488 | 0.78 |
| hot64k2 | 45.5 ms | 3,584 | 0.560 | 0.89 |
| hot64k8 | 47.7 ms | 2,048 | 0.344 | 0.55 |
| hot32k4a (added) | 21.7 ms | 4,096 | 0.631 | 1.05 (v2 in the same session 0.602) |
| hot64k4a (added) | 43.3 ms | 4,096 | 0.609 | 1.01 |
| hot96k4a (added) | 64.9 ms | 4,096 | 0.614 | 1.02 |
Reading: the verifier gets cheaper with `k`, because a hot load is one table read where a dataset load is an item derivation (9 mixers and 8 cache reads); the per-epoch cost is the fill, 24 to 73 ms on one core, against a 3,600 s epoch and the 20-minute seed lead. The 10 ms gate (spec 1.16) is unaffected.
### 6.4 The chip model with the measured g (M5 Max; the PC rows will replace it)
Chip A (no SRAM for `H`): gain after = `G0 / g`. Chip B (`H` in SRAM): `G0 x A_die / (A_die + A_S)` at 0.39 mm^2 per MiB, 100 mm^2 die.
| Class | g (M5 Max) | Chip A, gain after as a share of G0 | Chip B, share of G0 at 100 mm^2 | Chip B at 750 mm^2 |
|---|---|---|---|---|
| hot32k4 | 1.22 | 0.82 | 0.89 | 0.98 |
| hot64k4 | 1.12 | 0.89 | 0.80 | 0.97 |
| hot96k4 | 1.05 | 0.95 | 0.73 | 0.95 |
| hot64k8 | 1.71 | 0.58 | 0.80 | 0.97 |
The added form (second Mac session, 21:03 to 21:19 UTC, load average 7 to 14; v2 in that session 27.63 Metal, 27.59 OpenCL): the GPU keeps 0.93 / 0.87 / 0.83 of its rate at 32 / 64 / 96 MiB (`k = 4`), against the probe's 0.96 / 0.94 / 0.93, so the hot hits cost this card more than the probe says as the table grows, and 96 MiB is not resident. The named chip's position under the added form, same assumptions:
| Chip | Hot loads served from | Rate against its own version 2 rate | Gain after, as a share of G0 (S = 32 / 64 / 96) | Arithmetic |
|---|---|---|---|---|
| A, no SRAM for H | DRAM | at most 16 / 20 = 0.80 (20 dependent DRAM loads per iteration in place of 16) | 0.86 / 0.92 / 0.96 | `0.80 / g` |
| B, S MiB of SRAM for H, 100 mm^2 die, 0.39 mm^2/MiB | SRAM (the k reads near free) | 1.00 | 0.96 / 0.92 / 0.88 | `(1 / g) x A_die / (A_die + A_S)` |
| B on a 750 mm^2 die | SRAM | 1.00 | 1.06 / 1.12 / 1.15 | the SRAM is 1.6 to 4.8% of the die, the GPU's loss is larger |
With the PCs' `g` (added form, `k = 4`): RTX 5090 0.87 / 0.85 / 0.84, RX 9070 XT 0.84 / 0.81 / 0.80 at 32 / 64 / 96 MiB.
| Chip, added form, S = 32 / 64 / 96 MiB | Gain after as a share of G0, 5090 g | 9070 XT g |
|---|---|---|
| A, H from DRAM (rate 0.80 of its own) | 0.92 / 0.94 / 0.95 | 0.95 / 0.99 / 1.00 |
| B, S MiB of SRAM, 100 mm^2 die, 0.39 mm^2/MiB | 1.03 / 0.94 / 0.87 | 1.06 / 0.99 / 0.91 |
| B, 750 mm^2 die | 1.13 / 1.14 / 1.17 | 1.17 / 1.20 / 1.22 |
Reading: on the cards the plan named, the added form taxes no chip. A chip that serves H from its DRAM loses at most 8% of its gain on the 5090's figures and nothing on the 9070 XT's; a chip with the SRAM comes out ahead on every row except the smallest die at 64 and 96 MiB. The replaced form helps the recompute chip outright (section 2.3). The hot hits are not free on a GPU while the 1 GiB dataset streams through the same cache, and the layer's premise (section 1) needed them to be.
**Recommendation (owner: Josh, gate 1):** do not adopt layer 5 in either form on these measurements. What could change it: a GPU-side way to keep H resident (cache partitioning or persisting-access controls exist on NVIDIA, approximate, from memory, and are a driver setting, not a consensus rule), or a dataset access pattern that bypasses the cache; both are outside the hash and were not measured. The experiment stays behind the flag with its packs and vectors as the record.
Reading, replaced form: on the M5 Max figures the chip's cheaper way out of `hot64k8` is the SRAM (0.80 of `G0` at a 100 mm^2 die) rather than serving `H` from DRAM (0.58). The layer's value is then the SRAM area, which is small on a large die (0.97 at 750 mm^2). The strongest configuration on this card is the one whose table fits the cache and whose `k` is large; whether 64 MiB fits the 5090's L2 and the 9070 XT's Infinity Cache with the dataset streaming beside it is the PC measurement.
## 7. Decision and what would make it pay (the Counter ASIC 3.0 note)
Decided by the coordinator on 5 October 2026 after the PC rows: layer 5 is out of v3. The rule was a GPU cost under 3% (`g` above 0.97) for a chip cost worth having; measured `g` 0.84 to 0.87 on the RTX 5090 and 0.80 to 0.84 on the RX 9070 XT for the added form, and the replaced form helps the recompute chip. No re-run.
What would make a hot table pay, for a later round:
| Condition | What the measurements say | What would have to be shown |
|---|---|---|
| A table that stays resident beside a streaming 1 GiB | The probe's ceiling at every S inside the cache (5090: 112.6 G loads/s at 32, 64 and 96 MiB) and the hash's 1.02 to 1.08x say the dataset's random lines evict it; 32 MiB on the M5 Max kept 92% of its probe gain, the 5090 kept 29% | A size small enough to survive: the probe cannot say which (it has no competing stream); a sweep of `S` down from 32 MiB (16, 8, 4, 2) with the hash itself, on the 5090 first, would. A 2 to 4 MiB table costs a chip under 2 mm^2 of SRAM, so the layer would then tax nothing; the point of the layer was a table a chip cannot afford, and a table a GPU keeps is one a chip affords |
| The `k` and `S` the probe says | `g` moved with `k` (1.20x at `k = 8`, 1.02 to 1.08 at `k = 4`, 1.00 at `k = 2`) and barely with `S`; the probe predicted 1.27 to 1.73 | Only a resident table makes `k` worth raising; at `k = 8` with a resident table the replaced form reaches 2x (ideal) and the added form costs 0 by the probe. Without residency, `k` buys rate on the replaced form (which helps the chip) and costs rate on the added form |
| A different access shape | Both forms read H at one random word per hot load, the dataset's own pattern, and share the cache with 128 dataset reads per hash | A shape that touches the table in cache-line units and few lines per hash (one 64-byte line per iteration, say) would need a smaller resident set per hash and could be measured with the read-width emitters (`wide_load_stmt`) pointed at H. Or a hot table whose lines are read in a fixed order per epoch (a stream, not a random read), which a GPU prefetches and a chip must still hold or fetch; not designed here |
| Cache partitioning on the GPU | NVIDIA exposes persisting L2 access controls to CUDA programs (approximate, from memory); AMD and Apple do not expose an equivalent to OpenCL or Metal | A consensus rule cannot depend on a driver feature of one vendor; it could be a miner-side optimisation if the layer were in, which it is not |
What the round leaves in place: the code behind the flag (`LoadClass::hot`, both forms, the hosts filling H on the device from the epoch seed, the eight packs and their vectors), the probe at the hot sizes on three cards, and the measured rule that a read-only table a GPU cache could hold is not one it does hold while the dataset streams.
## 8. What is unverified
Listed here until measured, and carried into the bench-log entry.
1. Measured on all three cards (6.2): no card keeps `H` resident enough to deliver the probe's hit rate while the dataset streams. Not measured: whether a driver-side cache partition (NVIDIA's persisting L2 access controls, approximate, from memory) would; it is not something a consensus rule can rely on.
2. The PC rows are one run each (jobs run-ca2-hot-5090-20261005 and run-ca2-hot-9070-20261005, 5 October 2026); the two launch shapes per card agree within 1%, and no re-run was taken (PC 1 was handed on).
3. The resident warp counts of section 3 are approximate; the occupancy the hosts reach is printed by each harness (`kernel:` lines) and should replace them.
4. The SRAM area figures are approximate (section 4 cites their sources); no chip was priced.
5. `S = 96` uses the multiply-shift mapping; the spec's dataset still uses `AND MASK`. If the layer goes in, gate 1 decides whether the dataset mapping follows (section 1.13.3) or `S` stays a power of two.
6. The hot table on the chain needs the epoch seed about 1 ms (GPU) or up to 71 ms (CPU) before the epoch starts; the 20-minute lead of section 4.3 covers it. Not exercised on a node.
7. The Mac numbers were taken at load average 14 to 27 (other agents' CPU work); the ratios between packs of one session are the result, the absolute Mhash/s are not quiet numbers.

418
docs/plans/mixer-x4.md Normal file
View file

@ -0,0 +1,418 @@
# Mixer x4 and the cache growth rule: the class v3 dataset construction
5 October 2026 (night). Counter ASIC 2.0, layer 6 (option C) and ledger M16's lever, decided by the coordinator
under Josh's delegation at 22:00 UTC (`docs/plans/counter-asic-2-status.md`, "22:00 decided"; Josh confirms for
the public testnet genesis). Branch `ca2-mixer`. Worker: ca2-mixer (cryptographer's lane).
What this changes, in one line: under program class v3 every mixer application of the dataset item derivation
becomes four applications with distinct round keys, the eight dependent cache reads per item stay eight, and the
256 MiB cache doubles on the days the dataset doubles (years 4 and 12). Version 2 is byte-identical: the two
pinned packs re-export without a changed byte (section 5).
## 1. Why this form
The recompute attacker of `docs/analysis/m16-recompute-attacker-2026-10-05.md` holds the 256 MiB cache on a die
and derives every dataset word instead of reading it: 128 items per hash at about 1,170 integer operations and 8
dependent cache reads each. Its cost is linear in operations per item; the honest miner pays the mixer once a day
in the dataset build and never per hash; the verifier pays it per item it checks. The multiplier `m` is the one
parameter that moves the attacker and leaves the honest hash rate untouched.
Two shapes give the attacker 4x the operations:
| Shape | Mixer applications per item | Dependent cache reads per item | What grows for the verifier | What grows for the chip | What grows for the honest build |
|---|---|---|---|---|---|
| A, chosen: `m = 4` applications per round, 8 rounds | 36 | 8 | the ALU part only; the latency part (8 dependent misses per item) unchanged | integer operations 4x; SRAM bandwidth unchanged (1,024 reads per hash) | 4x the mixer arithmetic, same reads |
| B, alternative: 32 rounds of one application and one read | 33 | 32 | both parts: 4x the dependent misses per item, so about 4x the latency-bound time (spec 1.11: about 8 x 100 ns per item in series without interleaving) | integer operations 3.7x and SRAM bandwidth 4x (4,096 reads per hash, 60 TB/s to match one 5090 at the M16 rate) | 4x the reads too; the GPU build becomes latency-bound at 4x the dependent line fetches |
Shape A is chosen because the verifier's latency part is the part the 10 ms gate protects (section 1.11: the
distinct items of a unit are derived with their chains interleaved so the 8 misses of each item overlap across up
to 32 items; shape B would make that 32 misses deep). Shape B is implemented nowhere; the rule in the brief: if
shape A's measured verifier time exceeds 4.8 ms per warp on one M5 Max core, measure both and recommend. Section 6
has the measurement; it is under that bound, so B stays unimplemented.
## 2. Spec text (replaces 1.8.5 and 1.13.3 under class v3; v2 text unchanged)
### 1.8.5 Item derivation and dataset mapping
Item `t` (16 words) under mixer multiplier `m` (`m = 1` for program class v2, `m = 4` for class v3; a class
parameter, `LoadClass::mixer_mult`):
```
s[i] = K[i] for i in 0..7
s[8 + i] = t * MUL[i] + RC[i] for i in 0..7
for r in 0..7:
for j in 0..m-1:
s = M(s, rk = (r * m + j + 1) * 0x9E3779B9)
a = s[0] AND (2^(C - 4) - 1) cache line index, 2^(C - 4) lines of a 2^C-word cache
s[i] = s[i] XOR cache[line a][i] for i in 0..15
for j in 0..m-1:
s = M(s, rk = (8 * m + j + 1) * 0x9E3779B9)
item(t) = s
```
`M(s, rk)` is the mixer of 1.8.4 with round key `rk`; under `m = 1` the keys are `(r + 1) * 0x9E3779B9` and
`9 * 0x9E3779B9`, the version 2 text exactly. The round keys of the `9 m` applications are the first `9 m` values
of the version 2 key sequence, all distinct (the sequence is `k * 0x9E3779B9` for `k = 1 .. 9 m`, and
`0x9E3779B9` is odd, so no two of the first 2^32 keys coincide). Eight dependent cache reads per item at every
`m` (`ITEM_ROUNDS = 8`, prototype value): the address of read `r` depends on every earlier read. `9 m` mixer
applications, 36 under class v3, about 4,700 integer operations per item (130 per application, 1.8.4).
`dataset[w] = item(w >> 4)[w AND 15]`. A dataset of 2^D words is the prefix of items `0 .. 2^(D-4) - 1`, so an
item has the same value at every dataset size; and a cache of 2^C words is the prefix of segments of every
larger cache (1.8.3 fills segments independently of the cache size), but an item's value depends on `C` through
the line mask, so the item changes on the day the cache doubles.
Source: `igneum-pow/src/memhard.rs` (`derive_items`, `round_key_mult`, `Shape`), the emitted `mh_item` of
`memhard.h`, `memhard.metal` and `kernel.cl` (`emit.rs`, `emit_memhard_core`: the `m` loop is emitted only for
`m > 1`, so every version 2 pack keeps its text).
### 1.13.3 Dataset growth (class v3: option (b) with the cache tied to it, "option C")
Designed: 2 GiB at genesis plus 0.5 GiB per year. The linear schedule in bytes, `G x (1 + 86,400 d /
(4 x 31,536,000)) = G x (1 + d / 1,460)` for the genesis size `G` and the chain day `d` (DAA days since genesis,
section 1.12), doubles at day 1,460 (year 4), quadruples at day 4,380 (year 12), reaches 8x at day 10,220
(year 28). Rule (Designed, decided 5 October 2026 for class v3):
```
doublings(d) = floor(log2(1 + d / 1460)) integer division, then integer log2
dataset_words(d) = 2^(D_0 + doublings(d)) D_0 = 29 designed (2 GiB), 28 on the devnet (1 GiB); capped at 32
cache_words(d) = 2^(26 + doublings(d)) 256 MiB, 512 MiB from year 4, 1 GiB from year 12
```
Power-of-two sizes only (option (b)), so every load keeps the `src AND MASK` form of 1.14 item 2 and the cache
line index keeps `s[0] AND mask`. The cache doubles exactly when the dataset doubles ("option C",
`docs/analysis/sram-mirror.md` section 7): the cache's job is to stay above any GPU's last-level cache and that
needs growth; the recompute attacker is priced by the mixer, not by the cache (section 7 below).
`d` is `day_index(header.timestamp) - day_index(genesis.timestamp)` with `day_index = timestamp_ms / 86,400,000`
(`bind::day_index`, the interim day rule), clamped at 0 (`memhard::days_since_genesis`). Under class v2 nothing
grows: the cache is 2^26 words and the dataset the genesis size on every day.
| Chain day `d` | Years | `doublings` | Cache words | Cache | Dataset words (devnet `D_0 = 28`) | Dataset (designed `D_0 = 29`) | Verifier cache fill, one M5 Max core (measured at 256 MiB, section 6, scaled linearly) |
|---|---|---|---|---|---|---|---|
| 0 to 1,459 | 0 to 4 | 0 | 2^26 | 256 MiB | 2^28 (1 GiB) | 2 GiB | 0.18 s |
| 1,460 to 4,379 | 4 to 12 | 1 | 2^27 | 512 MiB | 2^29 (2 GiB) | 4 GiB | 0.36 s |
| 4,380 to 10,219 | 12 to 28 | 2 | 2^28 | 1 GiB | 2^30 (4 GiB) | 8 GiB | 0.72 s |
| 10,220 to 21,899 | 28 to 60 | 3 | 2^29 | 2 GiB | 2^31 (8 GiB) | 16 GiB | 1.4 s |
| 21,900 and on | 60 and on | 4 | 2^30 | 4 GiB | 2^32 (16 GiB, the index cap) | 2^32 words, the cap | 2.9 s |
Test: `memhard::tests::growth_schedule_table` pins every row and the day before each step. The devnet pack
`igneum-devnet-v4-epoch0` is day 20,730 of the Unix count against genesis day 20,729, `d = 1`, so every existing
size and vector stands.
Consequences for the tiers (the rule of 5 October): a verifier (any node, any pool core) holds 512 MiB from year 4
and 1 GiB from year 12, and fills it once a day in under a second on one 2026 core (the table); a miner's card
holds the dataset, 4 GiB from year 4 and 8 GiB from year 12 on the designed schedule, so an 8 GB card mines until
year 12 and a 16 GB card until year 28 (the cache is not in the card's working set at hash time: it is built,
the dataset built from it, and dropped). Those dates are the design document's own schedule restated as steps;
option (a) would have faded a 4 GiB card out in year 4 instead of year 4.
## 3. Interfaces
| Item | Where | Note |
|---|---|---|
| `LoadClass { mixer_mult: u8, growth: bool }`, `LoadClass::MX4` ("mx4"), `with_mixer(m, growth)`, `v2_loads()`, `takes_width_roll()` | `generator.rs` | a class with v2 loads takes no width roll: its program stream is version 2's draw for draw, so the v3 program of a seed is the v2 program of that seed, only the dataset differs |
| `Shape { mixer_mult, cache_log2_words }`, `Shape::for_class_day(class, d)`, `MixParams.shape`, `Cache::fill_log2(key, log2)`, `round_key_mult(r, j, m)` | `memhard.rs` | `Shape::V2` is version 2 |
| `growth_doublings(d)`, `cache_log2_words(d)`, `dataset_log2_words(D_0, d)`, `days_since_genesis(day, genesis_day)` | `memhard.rs` | the schedule, one function and its two sizes |
| `DatasetSource::{new_shape, from_key_shape, shape}`, `Epoch::new_class_day`, `Epoch::from_seed_bytes_day(epoch, day, label, class, d, D_0)` | `verify.rs` | the day-sized entries; the v2 entries are unchanged and build the v2 shape |
| `IGNEUM_MIXER_MULT`, `IGNEUM_CLASS_MIXER_MULT`, `IGNEUM_CACHE_GROWTH` in program.h; `"mixer_mult"`, `"cache_growth"`, the `"item"` string in program.json | `emit.rs` | written only for a class with `m != 1` or growth, so v2 packs do not change |
| `packfile.h` `mixerMult`; `packbench` and the OpenCL host print the multiplier and the cache size | the three hosts | the kernels carry the construction in their text (one emitter, three dialects); the hosts size the cache from `IGNEUM_CACHE_LOG2_WORDS` already (packbench.swift line 56, host.cu line 65, host.c line 1966) |
| `igneum-pow --class mx4 [--days d]` on every command | `main.rs` | `--days` sizes the cache for a growth class |
Under the ca2-v3 seam (`ProgramClass::V3`, `V3_CLASS`), the integration sets `V3_CLASS = LoadClass::MX4`; the
chain's day-sized dataset needs the day index, so `Epoch::chain_dataset(day, class)` builds the genesis-size
cache and a `chain_dataset_day(day_bytes, class, d, D_0)` beside it is the growth entry (section 9, owed to the
node agent).
## 4. Vectors (class v3, `proto-cuda/packs-ca2-mixer/`)
Produced by `igneum-pow export --program-class v3` (the Rust CPU interpreter, 5 October 2026, commit 66eeba3) and
checked on the GPUs in section 6. The program of each pack is the version 2 program of the same seed instruction
for instruction (`tests/packs.rs`, `v3_packs_are_the_v2_seeds_under_mixer_x4`); the cache is the version 2 cache
(day 0 of the growth rule); the dataset words and the hashes are new. The 64 sampled indices are those of every
pack (`emit::sample_indices`).
Pack `mx4-genesis` (seed igneum-genesis, day 2026-10-03, generator 3, class mx4, program id e323b9dcaf283a6f, 2^28 words, 2^26-word cache, cache FNV-1a 64 `48c4f5bf24166b2e` as under v2):
```
dataset words 0..15 (item 0):
61ff2180 0d4c7e6c 2177d443 60df9025 cf8b2e10 63675bfb 25289e58 9c45dc42
2d271c54 9652369b 2dd77508 5921392c 3afa60ee c640ad68 f2bb56ff cfa46438
dataset[0x0fffffff] = 5020180e
sampled words (the first 8 of the 64 in vectors.json):
dataset[59471966] = de85726d dataset[217795994] = 7cfc31c7 dataset[208353206] = d3cd5289 dataset[42483309] = 6ebbeef1
dataset[172547758] = 9858413b dataset[148076330] = e786f141 dataset[183853158] = 64f13833 dataset[214389424] = ca229d04
unit at base nonce 0, lanes 0..31:
63acd2d273f475ba e929c78b34b80d4b 0b1011cb19982558 1457a0df5497aa11
957d0f3bb71d98fb ac16901e6e6f6057 8ea1c6279f4b177a f28146e60bd08ba9
fe5b8cfe87f8e65b 49f87240566ace62 6ef6d6b7bdea8e41 46d9c0dc29a97b9c
111fe30128db9398 66dc39084f0946d4 8ee11bdfd35fecf2 2861fcfc75db6677
31c7667d4bde8556 c5989c48858b4ce0 276395e734a9d30d 84217b41e91368ff
3604861e34d9f697 9f51d8ee16bf3639 c89e47bafa84401c 7ae78c1f10b70e19
0b8c947157a29a48 d67192e8cfb43842 05a4c6d182c8c675 188e2661f3263f2e
a2df24238f7fea2e ed69ea7e13ad3a48 2d44ae509bab91b8 adad61931ea4fb70
unit at base nonce 4096: lane 0 edd508ac57e5699a, lane 1 4eefd56d526cdaeb, lane 31 8892f8604733b1e0
unit at base nonce 1000000: lane 0 8b3183778a49f59c, lane 1 1831b72a8797e895, lane 31 75eae55eba53a506
```
Pack `mx4-devnet-epoch0` (seed the devnet genesis hash, day bytes of 2026-10-04, generator 3, class mx4, program id 73bcbfe8ccf988f1, 2^28 words, 2^26-word cache, cache FNV-1a 64 `448274a57f508cbc` as under v2):
```
dataset words 0..15 (item 0):
afe80d67 b9fbd029 6c79f193 95139ad9 96310aff 4609f8b1 75279e63 28235be1
47b17dcb 718e0ef2 a52588c8 a8bf49d5 19cf243e 5ec8905e a4851f66 af9cd9f3
dataset[0x0fffffff] = e6a99c7a
sampled words (the first 8 of the 64 in vectors.json):
dataset[59471966] = 57642b58 dataset[217795994] = c279badd dataset[208353206] = cbccbaad dataset[42483309] = 32cce392
dataset[172547758] = 71fdb4c6 dataset[148076330] = f5a268ce dataset[183853158] = 3ca1d676 dataset[214389424] = 7977b03d
unit at base nonce 0, lanes 0..31:
212c6442b51e87ae c374795c00839331 b6036a220a98f4b3 eb8b8013e637367b
db5866e9b73930fd f3f3d01f46e90333 9d913991ab8ed428 7ccb1d8fa100a800
3cf45ba44f09a0fe 91acf48ef1a63082 6ea46c69fb082f99 581f0218977a9d72
9a4623a5c62ddf2d ab6eb5e768f0feb4 07b70bdccf8aca12 d666311ae5e4311e
53114757d669f0a4 bd5d6ace87ce2ce4 fb712015e8189192 a32cec81103e134b
83f3d18c3289c124 fe29f1984b132b3d c9ffcf4e3774497a ac99c9243dc63809
d78a7e8217a32f3c ab81ad63d242fc31 0e4c30b7e00024af ce014289fff6778d
64a1292e2a8b4a91 d5b8c90e681d7e3a 06078117673030fd 51bf77b280173930
unit at base nonce 4096: lane 0 3c797978566b5950, lane 1 7c759e60185b6411, lane 31 96a903eb9a0ca390
unit at base nonce 1000000: lane 0 d5a8da0568df8ee7, lane 1 68cb69c04208285a, lane 31 f8ca84a1a5d78cf5
```
## 5. The v2 path is byte-identical
`cargo test --test packs` regenerates every file of `igneum-genesis-mh` and `igneum-devnet-v4-epoch0` from
`program.json` and compares byte for byte (`emitted_sources_match_all_packs`, `export_pack_matches_all_packs`);
section 6 also records a fresh `igneum-pow export` of both packs diffed against the checked-in directories.
## 6. Measurements
Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0, 5 October 2026 (night), other agents' builds running beside every
run; a timing row says which lock it ran under (`measure` is exclusive; `run` and `build` are not timings).
### 6.1 The v2 path, byte for byte (no lock needed)
`igneum-pow export --seed igneum-genesis --day 2026-10-03` and `igneum-pow export --epoch-hex edc4fa84...fb07
--day-hex 69676e65756d2d6461792ffa50000000000000` on commit 66eeba3, `diff -r` against
`proto-cuda/packs/igneum-genesis-mh` and `igneum-devnet-v4-epoch0`: IDENTICAL, both (twelve files each). The
crate tests regenerate the same files and compare them on every run (`tests/packs.rs`, 12 of 12 pass).
### 6.2 Bit-exactness of the class v3 construction on the GPUs (`with-lock.sh run`, 22:05 UTC)
`packbench --pack <dir> --batches 1 --batch-log2 24 --group 256` (Metal, built from this branch) and
`igneum-bench-cl-igneum-genesis-mh --bench-pack --pack <dir> --batches 1 --batch-log2 24` (Apple OpenCL, built from
this branch's host.c). Vectors are the Rust interpreter's; the fingerprint is FNV-1a 64 over the 2^24 outputs at
base nonce 0.
| Pack | Harness | Cache FNV-1a 64 | Dataset head, word MASK, 64 samples | Vectors standalone / in batch | Fingerprint 2^24 | MH/s (GPU time; not a measurement, the run lock) |
|---|---|---|---|---|---|---|
| mx4-genesis | Metal | 48c4f5bf24166b2e PASS | PASS | 3/3, 3/3 | 6f48d5a2aa0dbe5f | 27.6 |
| mx4-genesis | Apple OpenCL | PASS | PASS | 96 of 96 lanes | 6f48d5a2aa0dbe5f | 27.7 (wall) |
| mx4-devnet-epoch0 | Metal | 448274a57f508cbc PASS | PASS | 3/3, 3/3 | 73caaebb28e808fe | 27.5 |
| mx4-devnet-epoch0 | Apple OpenCL | PASS | PASS | 96 of 96 lanes | 73caaebb28e808fe | 27.6 (wall) |
| mx8-genesis (the x8 candidate, 21:45 UTC) | Metal | 48c4f5bf24166b2e PASS | PASS | 3/3, 3/3 | 7c28cfb06c5c65a9 | 27.7 |
| mx8-genesis | Apple OpenCL | PASS | PASS | 96 of 96 lanes | 7c28cfb06c5c65a9 | 27.6 (wall) |
| mx8-devnet-epoch0 | Metal | 448274a57f508cbc PASS | PASS | 3/3, 3/3 | bbb183f72692f840 | 27.7 |
| mx8-devnet-epoch0 | Apple OpenCL | PASS | PASS | 96 of 96 lanes | bbb183f72692f840 | 27.7 (wall) |
Reading: the Rust interpreter, Metal and Apple OpenCL agree on the v3 dataset (head, MASK word, 64 samples), on
every vector lane and on the 2^24-output fingerprint of each pack; the hash rate is the v2 rate (27.7 MH/s on this
card tonight, readwidth table), as it must be: the hash kernel only loads, the mixer is paid in the build.
Dataset build under the run lock (indicative only; the measured rows are in 6.4): Metal 30.2 ms GPU (mx4-genesis)
and 21.7 ms GPU (mx4-devnet-epoch0) for 1 GiB; Apple OpenCL 55 and 56 ms wall. The readwidth entry's v2 figure on
this card is 0.6 to 2.1 ms cache fill and a 1 GiB build the bench-log's memory-hard entry puts at 13 to 30 ms; the
x4 build on the GPU is the row the measure lock will settle.
### 6.2a The packs as pinned after the x8 decision (`with-lock.sh run`, 22:12 UTC, the fixed binary's exports)
| Pack | How exported | Program id | Unit 0 lane 0 | Fingerprint 2^24 (Metal = Apple OpenCL) | Vectors, self-tests |
|---|---|---|---|---|---|
| mx8-genesis | `--program-class v3` (generator 3, V3_CLASS = mx8) | e323b9dcaf283a6f | 19b56348bc85304d | 7c28cfb06c5c65a9 | 3/3 + 3/3, 96 of 96, PASS |
| mx8-devnet-epoch0 | the chain path, `--program-class v3 --era-hex <genesis>` (the era drawn inside the class, load class `mx8-erad810f22d`) | 73bcbfe8ccf988f1 | d424577fce4a7a60 | 90f794dd556f7a3b | 3/3 + 3/3, 96 of 96, PASS |
| mx4-genesis (the x4 record) | `--class mx4` (generator 2, the class in the id) | 951b89584750bd75 | 63acd2d273f475ba | 6f48d5a2aa0dbe5f | 3/3 + 3/3, 96 of 96, PASS |
| mx4-devnet-epoch0 (the x4 record) | `--class mx4` on the chain seeds | (program.json) | 212c6442b51e87ae | 73caaebb28e808fe | 3/3 + 3/3, 96 of 96, PASS |
The PC 1 rows of 6.5 ran the earlier mx8-devnet-epoch0 (the load-class export, fingerprint bbb183f72692f840, era
not inside); the composed pack's PC fingerprints are owed with the next PC round (section 9).
### 6.3 Soundness suite on the v3 construction (commit 66eeba3 and after; `with-lock.sh build` for cargo, `run` for the GPU)
| Suite | Command | Result |
|---|---|---|
| Crate lib tests (memhard schedule table, the by-hand multiplied mixer, the seam, the generator) | `cargo test -j4 --release` | 44 of 44 pass |
| Pinned packs, v2 and v3 (`tests/packs.rs`: programs, ids, dataset words, 96 vectors per pack, every emitted file byte for byte, 16 masked loads per kernel, the v3 packs as the v2 seeds under mixer x4) | same | 12 of 12 pass |
| Scratch soundness tests of branch ca2-soundness (cherry-pick 0d8f745, one conflict in packbench.swift's RESULT line resolved by hand, `era_bytes: None` added to the edge program literal) | `cargo test -j4 --release --test scratch` | 7 of 7 pass: bijections, re-hit rates, 56 of 56 edge units, 42 of 42 emitted scr kernels, 200 scratch programs and 800 units on the CPU |
| v3 fuzz on the CPU (`tests/mixer.rs`): 200 programs through the seam, the contract on every instruction (each is the v2 program of its seed), 4 units each across the 32-bit range with one unit in the top 256 nonces, interpreted twice | `IGNEUM_MIXER_PACKS_OUT=<dir> cargo test -j4 --release --test mixer` | 200 of 200, 800 of 800 units; 200 packs written for the GPU runs (82 s with the three cache fills) |
| v3 stats beside v2 (`tests/mixer.rs`): 8,192 outputs per seed, bit balance, single-bit avalanche within the unit and across units, duplicates | same | igneum-genesis v3: avalanche 49.99 percent, worst bit z 1.92, 0 duplicates (v2: 49.87, z 2.25); igneum-genesis/stats1 v3: 49.97, z 3.09 (v2: 49.98, z 2.30) |
| v3 edge (`tests/mixer.rs`): items 0, 1, 2^28 - 1 and 2^32 - 1 by hand at m = 1, 2, 4, 8 on a 2^14-word cache; words 0, 15, 16, 17, MASK - 1, MASK through the interpreter's fetch path; the index wrap at MASK + 1 | same | pass |
| v3 determinism (`tests/mixer.rs`): two independent epochs, every vector and every emitted file equal, and equal to the pinned pack | same | pass |
| Metal and Apple OpenCL on the two pinned v3 packs | section 6.2 | 3/3 standalone, 3/3 in batch, 96 of 96 lanes, dataset head, MASK word and 64 samples, one fingerprint per pack across both harnesses |
| Metal fuzz: the 200 packs, 4 units each standalone and the top-256 unit inside a 512-nonce batch at base 4,294,967,040; every tenth pack on Apple OpenCL as well | `packbench --pack <dir> --batches 1 --batch-log2 9 --batch-base 4294967040` (Metal), `igneum-bench-cl-igneum-genesis-mh --bench-pack --pack <dir> --batches 1 --batch-log2 10` (Apple OpenCL), `with-lock.sh run`, 21:22 to 21:24 UTC | Metal 200 of 200 packs PASS (800 of 800 standalone units, 200 of 200 inside the wrapping window, cache and dataset self-tests on every pack); Apple OpenCL 20 of 20 packs PASS (the three vectors.h units, the self-tests); a first run with a packbench built before the `--batch-base` cherry-pick reported 200 of 200 FAIL on an empty RESULT line and was read as such (the watcher rule), the harness rebuilt and the run repeated |
### 6.4 Timings (`with-lock.sh measure`, one session, 21:40:12 to 21:40:23 UTC, commit 504cae4)
Script `measure-v3.sh` (session scratchpad): `igneum-pow bench --seed igneum-genesis --day 2026-10-03 --warps 50`
(v2), `... --program-class v3` (x4), `... --class mx8` (x8), two rounds each, then the devnet seeds, then
`packbench --pack <dir> --batches 2 --batch-log2 22 --group 256` on igneum-genesis-mh, mx4-genesis and mx8-genesis,
two rounds. The lock was exclusive among the agents' builds and measurements, but the box was not quiet: load
average 5.6 (one minute) and 26 (fifteen minutes) at the start, from unlocked processes (the devnet node, other
agents' editors); the v2 row reads 1.31 to 1.36 ms where the quiet readwidth night read 0.604 to 0.626. So the
absolute numbers below are a loaded-core figure, about 2.2x the quiet one, and the ratios between the rows are the
measurement (two rounds within 4 percent). A quiet-box re-run is owed (section 9).
| Construction | Verifier, ms per 32-lane unit, avg of 50 (round 1 / round 2) | Worst cold unit of three | Against v2 | 256 MiB cache fill, one core | Metal 1 GiB dataset build, GPU ms (round 1 / round 2) |
|---|---|---|---|---|---|
| v2 (igneum-genesis) | 1.361 / 1.310 | 1.579 | 1 | 172.1 / 172.6 ms | 29.7 / 21.0 |
| x4 (mx4, class v3) | 1.956 / 1.923 | 2.043 | 1.45x | 175.3 / 172.3 ms | 20.9 / 21.0 |
| x8 (mx8) | 2.785 / 2.790 | 2.942 | 2.09x | 172.3 / 172.3 ms | 21.9 / 21.9 |
| x4, the devnet seeds (mx4-devnet-epoch0) | 1.923 | 2.012 | | 173.9 ms | |
| x8, the devnet seeds | 2.972 | 2.885 | | 173.6 ms | |
Reading. The verifier's ALU part is what grows: x4 adds 0.6 ms per unit for 27 more mixer applications on each of
4,096 items (110,592 applications, about 5.5 ns each on this core, the lanes' chains interleaved), x8 another
0.85 ms for 36 more; the latency part (8 dependent misses per item) is the same in every row, which is why the
measured ratios are 1.45x and 2.1x and not the 4x and 8x of the M16 table's scaling. Shape B (32 rounds of one
read, section 1) would have multiplied the latency part too; x4 is under the 4.8 ms bar even on the loaded core, so
B stays unimplemented. The cache fill does not depend on the mixer (it is the ChaCha chain): 172 to 175 ms, the
spec's 175 to 181 ms of 1.8.3. The Metal 1 GiB build does not move with the mixer at all (21 ms at v2, x4 and x8
once warm; the 29.7 ms first v2 run is the first-touch cost the hosts fill twice for): on this card the build is
bound by the 8 dependent cache-line reads per item, not by the arithmetic, so the Mac says nothing about whether
the 5090's or the 9070 XT's build is arithmetic-bound; that is the PC job (section 8).
Against the x4 / x8 rule (section 6.5): the verifier half passes for x8 with 7.1 ms of the 10 ms gate to spare on
this loaded core (worst cold 2.94 ms; the quiet-core figure would be about 1.3 ms, scaled by the 2.2x of the v2
row, approximate); x4 leaves 8.0 ms. The build half waits on the PC rows.
### 6.5 Verification throughput per tier (consequences row C19), from the loaded-core figures above
Warps verified per second on one core = 1,000 / (ms per warp); a pool core verifying members' shares handles that
many shares per second; a node verifies a block with one unit (plus the header path, under 0.1 ms, not measured
here); IBD over the 108,000-header pruning window (spec 02) on one core = 108,000 x ms per warp.
| Figure | v2 | x4 | x8 | Note |
|---|---|---|---|---|
| ms per warp, steady (this session, loaded core) | 1.33 | 1.94 | 2.79 | avg of the two rounds |
| ms per warp, quiet M5 Max core (scaled by 0.604 / 1.33 = 0.45, approximate) | 0.60 | 0.88 | 1.26 | the readwidth night's v2 figure is measured; x4 and x8 scaled |
| ms per warp, 2019-class laptop core (approximate: 2.5x the quiet M5 Max figure, the ratio the design document assumes for the gate; unmeasured, O-1.14) | 1.5 | 2.2 | 3.2 | the figure that fixes the gate is a measurement, not this row |
| Shares per second per core (loaded / quiet, approximate) | 750 / 1,660 | 515 / 1,140 | 358 / 790 | spec 09 section 9.8 item 5 carried 2,270 at v2; re-cut from the quiet row: 1,660 |
| Cores for a 22,000-member pool at one share per member per 10 s (2,200 shares per second), loaded / quiet | 2.9 / 1.3 | 4.3 / 1.9 | 6.1 / 2.8 | |
| Node: worst cold single unit (loaded core) | 1.58 ms | 2.04 ms | 2.94 ms | per block |
| IBD over 108,000 headers on one core, loaded / quiet, minutes | 2.4 / 1.1 | 3.5 / 1.6 | 5.0 / 2.3 | laptop (approximate): 2.7 / 4.0 / 5.8 min; a seed VM core (unmeasured) sits between the laptop and the quiet M5 Max |
| Margin left under the 10 ms gate for Counter ASIC 3.0 (worst cold, loaded core) | 8.4 ms | 8.0 ms | 7.1 ms | on the 2019-class laptop row (approximate) 7.5 / 6.8 / 5.9 ms steady |
Reading: at x4 a pool core serves about 1,100 shares per second on a quiet 2026 core (a 22,000-member pool needs
two cores); at x8 about 800 (three cores). A node's block verification stays a few milliseconds. The gate's
remaining margin is what Counter ASIC 3.0 has to spend, and on the unmeasured laptop core it is 6 to 7 ms at x4 and
about 6 at x8, which is the number the 2019-class measurement (O-1.14) must confirm before x8 is final.
### 6.4a The same session on the fixed binary (section 6.6), `with-lock.sh measure`, 22:06:59 to 22:07:06 UTC
The verifier of section 6.4 was measured on a binary that carried the inlining regression of section 6.6; after the
fix, readwidth's binary and the fixed one on the same v2 input in the same minute (checksum 19297e99c7b9a55e), then
x4 and x8 on the fixed binary; load average 5.5 (the same box, so the ratios of 6.4 stand and the absolute row is
now the measured one):
| Construction | Verifier, ms per unit, avg of 50 (round 1 / round 2) | Worst cold unit of three | Against v2 |
|---|---|---|---|
| v2, readwidth e752fc7's binary | 0.607 / 0.610 | 0.666 | 1 |
| v2, the fixed binary | 0.609 / 0.611 | 0.657 | 1.00 |
| x4 (mx4) | 1.238 / 1.237 | 1.396 | 2.03x |
| x8 (mx8, class v3) | 2.077 / 2.058 | 2.145 | 3.4x |
| x4, the devnet seeds | 1.240 | 1.289 | |
| x8, the devnet seeds | 2.058 | 2.144 | |
The added cost per unit is the same as on the slow binary (x4 + 0.63 ms, x8 + 1.46 ms: the regression was a
constant 0.72 ms per unit in the shared item loop), so the ALU reading of 6.4 holds; the ratios against v2 are
2.0x and 3.4x once v2 is back at 0.61. The 10 ms gate keeps 7.9 ms at x8 (worst cold 2.15 ms) on this core.
### 6.5 The daily build per tier, and the x4 / x8 rule
The coordinator's rule (21:30 UTC): x8 enters v3 if the per-warp verify stays under 10 ms on one Mac core AND the
daily 1 GiB build stays under 1 s on every discrete card we own; else x4 with the thin margin stated and x8 named
as the next lever. The integrated tier is decided beside it (consequences row C23): its build is per prepare, not
per day, so its consequence is per-day dataset reuse in the workers or a restart per epoch.
| Card | Build at x1 | x4 | x8 | Source |
|---|---|---|---|---|
| RTX 5090 (PC 1, job run-mixer-x4-pc1-20261005, 22:00 to 22:04 UTC, the worker's `cache ... dataset ... ms` wall line, two dispatches per pack) | 23 to 25 ms | 23 to 25 ms | 23 ms | the PC 1 job (13.4 ms GPU time on 3 October: the wall line carries the launch) |
| RX 9070 XT (PC 1, gfx1201 on the eGPU, the same job) | 74 ms | 73 to 77 ms | 72 to 76 ms | the PC 1 job |
| M5 Max, Metal | 13 to 30 ms (the two runs of the memory-hard entry; tonight's run-lock figures 21.7 to 30.2 ms at x4 and 22.0 to 30.0 at x8 say the Mac's build is latency-bound, not mixer-bound) | section 6.4 | section 6.4 | this file |
| Radeon integrated gfx1036 (PC 2), OpenCL, per prepare | 6.9 / 9.4 / 11.7 s prepare total with the 1 GiB build inside | about 28 to 47 s (approximate: scaled x4; the iGPU's build is arithmetic-bound at x1 already) | about 55 to 94 s (approximate) | `docs/plans/epoch-length.md` section 7 (branch ca2-epoch), M11 table |
| gfx1036 beside WSL build jobs (PC 1) | 55 / 116 / 124 s | about 4 to 8 min (approximate) | about 7 to 17 min (approximate) | same |
| 8 GB-class discrete card (not owned; about a tenth of the 5090's rate, approximate) | about 0.13 s | about 0.5 s | about 1 s, on the edge of the rule | scaled from the 5090 row, approximate |
Decision (coordinator under the delegated rule, 22:05 UTC, on these rows): x8 enters class v3. Both halves pass:
the verifier at x8 is 2.1 ms per unit on one M5 Max core (6.4a) against the 10 ms gate, and the daily 1 GiB build
does not move with the mixer on any discrete card we own (5090 23 to 25 ms, 9070 XT 72 to 77 ms, M5 Max 21 ms at
x1, x4 and x8: latency-bound), 13x to 40x under the 1 s bar. `V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }`;
the pinned v3 packs are mx8-genesis and mx8-devnet-epoch0 (the latter through the chain path with the era inside the
class); the x4 packs stay pinned as the candidate's record (generator 2, the class in the id).
Reading of the integrated tier: on the discrete cards the rule is settled by the 5090 and 9070 XT rows above. The integrated tier misses the rule at x4 already: a per-prepare build of 28 to 47 s is a tenth to a quarter
of the 600-DAA-second lead the devnet gives the next program (spec 1.12), and under load it is the whole lead; so
if x4 or x8 goes in, the iGPU tier needs the workers to build the day's dataset once a day and keep it across
epochs (today a prepare rebuilds it: `proto-opencl/host.c` prepareTask builds the pair's cache and dataset per
prepare), or to restart per epoch. The node agent is asked whether the per-day reuse is bounded tonight
(coordinator, 21:45 UTC); until then the iGPU consequence stands as written.
### 6.6 The verifier regression of 0fc0ad1, found and fixed (5 October 2026, 21:59 to 22:07 UTC)
The era agent measured the same v2 input with two binaries in one minute: readwidth's 0.604 ms per unit, ca2-v3
HEAD's 1.33. Bisected under the measure lock (one session, four binaries, two rounds, checksum 19297e99c7b9a55e):
readwidth e752fc7 0.607 / 0.609; the ca2-v3 seam 6c75dad (before this branch) 0.610 / 0.609; this branch's 0fc0ad1
1.332 / 1.316; ca2-v3 88dafbc 1.325 / 1.347. So the 2.2x was in 0fc0ad1's `derive_items`, on the version 2 path the
devnet verifies with, and section 6.4's "loaded box" reading was wrong: the load was real (the same session shows
it) but the 2x was the code.
Variants, each a one-change copy measured against readwidth's binary in the same session:
| Variant | ms per unit | Reading |
|---|---|---|
| 0fc0ad1 as written (the line mask read from the cache at run time, the loop inlined into `MemhardCpu::fetch`) | 1.33 | the regression |
| the mask hoisted into a local before the item loop | 1.32 to 1.37 | not the reload |
| `Cache::line` with the constant mask, on the 0fc0ad1 tree | 0.60 to 0.65 | fixed there |
| the same constant mask on the merged ca2-v3 tree | 1.33 to 1.46 | not the mask either |
| one instance per cache size with the mask a constant, `#[inline(always)]` | 1.32 to 1.39 | not the mask |
| the same instances `#[inline(never)]` | 0.604 / 0.617 / 0.618 / 0.624 | the fix |
So it is inlining: the item loop inlined into its callers (`fetch`, `fetch_wide`, `word_at`) runs at 2.2x the
out-of-line loop, and which small change tips LLVM's decision depends on the rest of the tree (the constant mask
tipped it on one tree and not on the other). The fix (`memhard::derive_items`): the loop is `derive_items_mask`,
`#[inline(never)]`, one instance per cache size the growth rule reaches (2^26 to 2^30 words) with the line mask a
constant, and a run-time-mask instance for every other size (tests). Measured in 6.4a: 0.609 / 0.611 against
readwidth's 0.607 / 0.610.
The class, not the instance: a verifier benchmark with a pinned bound in the crate's CI (the v2 unit at a known
input against a stored ms-per-unit on a named core, failing on a 1.3x drift) would have caught this at the first
commit; filed for the next cut (section 9). Until then the era agent's two-binary check (same input, same minute)
is the rule for every change that touches the item loop.
## 7. The chip model
`docs/analysis/chip-model-v3.md`.
## 8. What is unverified
1. The 5090's and the 9070 XT's dataset build at x4 and x8 are measured (6.5, the PC 1 job): both latency-bound,
under 0.1 s. The gfx1036 is the tier that fails the per-prepare build (6.5), and its consequence (per-day dataset
reuse in the workers) is with the node agent.
2. The absolute verifier figures were taken on a loaded core (load average 5.6); the ratios are the measurement
and the quiet-core figures are scaled. A 2019-class laptop core has not run any construction (O-1.14).
3. The mixer has had no cryptanalysis (spec 1.8.4); `m` applications with distinct round keys is `m` times the
work only if no shortcut composes them, which is the same open question as for one application.
4. The x8 packs are generator 2 with the class in the id (`--class mx8`); if x8 is chosen, the pinned v3 packs are
re-cut through the seam (`V3_CLASS = MX8`, generator 3) and the tests re-pinned, one commit.
5. The 5090's rate for the v3 program is the v2 rate by construction (the hash kernel is unchanged, the Mac shows
27.7 MH/s at v2, x4 and x8); the chip row's denominator stays the readwidth table's 136.1 MH/s until a v3 pack
runs on the card, which the PC job also gives.
## 9. Owed
| Item | Owner | When |
|---|---|---|
| PC 1 run of the five packs | done 22:04 UTC (run-mixer-x4-pc1-20261005; 6.5 and 6.2a) | |
| A verifier benchmark with a pinned bound in the crate's CI (section 6.6, the class rule) | ca2-mixer | the next cut |
| GPU bit-exactness of the re-exported mx8-devnet-epoch0 (the composed class, the era inside) and the mx4 record packs on the PCs (the Mac rows are in 6.2a) | ca2-mixer | the next PC round |
| The x4 / x8 choice recorded from the rule, then the vectors re-cut once through the seam | done 22:05 UTC: x8 (6.5), V3_CLASS = MX8, mx8 packs re-exported | |
| `Epoch::chain_dataset_day` wired to the genesis day index in the node (`days_since_genesis(day_index(header), day_index(genesis))`) and `pow_genesis_dataset_log2` in the override | ca2-node | the integration |
| The spec text of section 2 into `docs/spec/01-lottery-hash.md` 1.8.5 and 1.13.3 (with the v3 vectors into 1.17) | the integration | after the choice |
| The 2019-class laptop core measurement that fixes the gate (O-1.14) | cryptographer | gate 1 |

143
docs/plans/proving-v1.md Normal file
View file

@ -0,0 +1,143 @@
# Proving v1: segment records, the chain rule, the unproven rule; the 0.3.11 rollout
5 October 2026, from 18:55 UTC (Josh: "open the proving round asap"). Branches `proving-v1` in the main repository
(worktree `/Users/joshm/Projects/igneum-wt-proving-v1`, from master a93199a) and in the fork
(`vendor/igneum-node-pv1`, from release-0.3.6 a24ab01a; to be rebased onto the 0.3.10 tip when it lands on
release-0.3.6). Status words follow `docs/spec/00-overview.md` 0.2. Every number here is in `docs/bench-log.md`
with its command. Nothing ships from this plan: it delivers branches, numbers and the rollout for 0.3.11.
## The gap this closes
The litepaper says every block is proven within about a minute. Proving v0 (`proving-v0.md`, spec 7.7) proves
some shards: one prover (PC 2's RTX 5090) takes the newest shard assigned to it, about one shard every 30 s, so
under a tenth of blocks carry a proof; consensus does not require one; the aggregator guest (design 5.3) runs
on fixtures only. Proving v1 adds the aggregated segment record on chain (spec 7.8), the chain rule (segment N's
record verifies N-1, inside the proof by recursion), the unproven rule (a segment nobody proves in T seconds pays
nothing and may be skipped), the prover on by default on every machine that can prove, and the measurements
that say how many cards cover the chain.
## The round, step by step
| Step | What | State |
|---|---|---|
| 1 | Prover on by default (`app/igneum-app/src/provedefault.rs`, `engine.rs apply_prove_default`): on at install when the machine can prove (NVIDIA card with 12 GB or more; WSL2 answering on Windows; Linux native; Apple silicon off until measured), never switching an explicit on back off; the Settings switch line and the tile line say why. Unit tests (5). The cost of proving on a mining machine: PC 2 job `prover-cost-pc2-pv1` (5 min mining alone, 5 min with the prover, the GPU memory peak and the host RAM peak, the sp1-gpu-server's compiled SM targets) | Implemented; the measurement is HELD (coordinator, 19:00Z): PC 2's RTX 5090 worker has been exiting on a pack seed mismatch since 18:35Z, so the first run's "mining alone" is 0 MH/s and void; re-run after the go |
| 2 | Segment aggregation: `SegmentRecord` (586 bytes, the aggregator guest's 340-byte statement inline), section `IGNS` before the shard section, p2p message 75 at protocol 15 (14 went to the EVM transaction relay in 0.3.10), the native block statement and the veto, the credit split (`split_pool_credit`), the payout at the carrier, `igneum-miner sign-segment-record`, the RPCs; host modes `chain` (consecutive fixtures), `aggregate` (live shard proofs from the pool, a run of blocks in one process) and `verify-segment` (the node's verifier, pinned aggregator key); the app's aggregator step (`prover.rs aggregate_once`) | Implemented, unit-tested (consensus core 2 new tests, exec 2, params 1); the GPU measurement (N = 2, 4, 8 blocks on PC 2) is HELD with step 1; the Mac CPU run of `--mode chain` over 2 live blocks is the known-finished case |
| 3 | Coverage: `tools/proving-v1/coverage.mjs` (the proven-block share and the on-chain proof latency over a window from one node's RPC, the live page beside it) | Implemented and run for 3 min (below); the 30-min window waits for the fleet |
| 4 | The chain rule and the unproven rule in consensus behind `proving_v1_activation_daa` (spec 7.8 items 2, 6, 7); unit tests; the fast-time 3-node harness `tools/proving-v1/net.mjs` (ports 29950+, suffix 956, trust mode) with the known-finished and known-failed cases | Implemented; the harness run waits for the Mac build of the fork (`vendor/igneum-node/target-pv1`) |
| 5 | This plan: the rollout for 0.3.11 and Josh's decisions | Written below |
## Numbers (every one from `docs/bench-log.md`, "proving v1: segment records ...", 5 October 2026 evening)
| What | Number |
|---|---|
| The prover's cost to a mining 5090 (the re-run with the fleet mining, hash rate from the miner's own STATUS lines) | 124.7 MH/s alone, 119.7 MH/s with the prover on: 5.0 MH/s, 4.0%, on empty shards at 1.4 a minute |
| GPU memory on the 5090: the prover alone (empty shards) / the miner and the prover together | max 13,816 MiB / max 15,590 MiB (the miner holds 3,396 MiB); a 24 GB 4090 has 8.4 GB of headroom, a 16 GB card 0.4 GB, a 12 GB card cannot do both on this build; the full-shard peak is the chain job's row |
| Coverage, 30-min window with the fleet mining, one prover | 2.4% of blocks, latency p50 44 s, p99 52 s |
| Host RAM | host used 25.6 GB of 63 GB; the WSL2 VM 7.9 GB working set |
| sp1-gpu-server 6.8.1 compiled targets (cuobjdump) | sm_80, sm_86, sm_89, sm_90, sm_100, sm_120 and compute_120 PTX: Ada (4090) is native, no JIT; nothing for AMD |
| Shards a minute, one 5090 through the app's loop (empty shards) | 1.6 |
| Chain of 2 live blocks on the Mac CPU (`--mode chain`) | shard 55.4 and 41.3 s, aggregate 52.0 s then 59.1 s with the previous proof, chain_len 2, final proof 1,272,909 bytes, `verify-segment` 0.032 s |
| Unit tests | consensus core 13, exec 8, app 5, all passing on the Mac |
| The harness (3 nodes, fast time, trust mode) | PASSED, 21 checks in 197 s on b177718e (N = 4) and 244 s on the final tree ece42979 (N = 8): paid 1.0 s after submit, every node agreeing; the fresh chain refused after a proven segment; the unproven segment skipped after its deadline; shards at 90% |
| Coverage, 3-min window, the degraded fleet (one card, the Mac verifier down) | 4.7% of blocks proven, on-chain latency p50 39 s |
| The chain of 8 live blocks on the 5090, the card also mining (`chain-pc2-pv1c`) | shard 7.3 to 7.7 s, first aggregation 7.9 s, every chained one 9.6 to 9.7 s; N = 2 in 32.6 s, N = 4 in 66.8 s, N = 8 in 135.6 s (17.0 s a block); the final proof 1,272,909 bytes whatever N, the record 586 bytes, `verify-segment` 0.037 to 0.040 s; GPU peak 16,751 MiB with the miner resident. Against 4 October with the miner stopped (aggregate 2.2 s): the miner slows the prover 3 to 4x |
| 5090-class cards for 100% at 1 block/s, measured rows | 47 with the loop as it is, 18 through the chain mode on mining cards, 6 (approximate) on proving-only cards, at empty blocks; 14 proving-only at one full shard a block; 45 at `B_p`: the table in the bench log |
## The rule, in one paragraph (spec 7.8)
From the first chain block `A` at or above `proving_v1_activation_daa`, chain blocks form fixed segments of `N`
(`proving_v1_segment_blocks`). A segment record carries the aggregated proof of the segment's last block, whose
`chain_len` says how many consecutive blocks the recursion attests. Every node checks the record's statement
against its own native block statement (every field but the provers commitment and `chain_len`), the chain rule
(a proof that does not chain to the previous segment, `chain_len = N`, is valid only for the first segment or
after an unproven one) and the deadline (`T = proving_v1_unproven_daa` DAA seconds after the segment's last block;
a record carried later pays nothing). The pool credit of every attested block splits: `proving_v1_aggregator_share_bps`
to the aggregator, the rest to the shards as v0. A block is never invalid for lack of a proof; the mandatory rule
(spec 7.8 item 10) is Designed and off, with no switch yet.
## Rollout for 0.3.11 (the digest handshake pattern of 0.3.9 and tonight's switches)
The switch moves the consensus digest only once it is set (`consensus_digest`: the four v1 fields enter the hash
when `proving_v1_activation_daa != never`), so a 0.3.11 node on the unswitched devnet keeps the 0.3.10 digest and
the rolling upgrade does not partition the network. The order, each step with its check:
1. **Rebase and build.** Fork `proving-v1` rebased onto the 0.3.10 tip on `release-0.3.6`; the six node suites and
the app tests as PC 2 build jobs; the Mac node and the Windows exes by the Mac cross-build; the Linux node by
PC 1; the HiveOS package republished from the same fork commit (rule: the HiveOS package carries the node of
the release commit and the same override object, `infra/hive`, as 0.3.9's `hive-sync-039o` checked it).
2. **Pinned guests.** The guest ids do not change in this round (shard `0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a`,
aggregator `0x474678f35f7545db28055d5e5bbc308231d84a5a072202087a2a8d5b09123896`, pinned 2026-10-05T16:20:38Z):
the aggregator guest already carried the chain rule and only the HOST gained modes. So no provers-off drain is
needed for the guests; `--mode id` on every machine after the update must print the same two ids, and the node
now reads them at start (`program_ids`: `IGNEUM_PROOF_PROGRAM_IDS` or the verifier's `--mode id`) and names them
in the native statement.
3. **Hand nodes and the seed first**, with the UNCHANGED override object (the digest stays): observer, node 1, the
seed on the 0.3.11 node; peers back within 20 s; the `proving v1: segment records from DAA score never` line in
each log.
4. **Manifest and apps.** `publish-manifest.sh --version 0.3.11` with the unchanged object; `update-now` to every
app; every machine on 0.3.11 with a DAA score and a hash rate (the watcher takes the commit as an argument).
The app's prover default applies at the first start on 0.3.11: every NVIDIA machine with WSL2 goes on; the
log line `prover default: ...` on each.
5. **The switch.** When every node runs 0.3.11: publish the object with `proving_v1_activation_daa` = H (24 h
ahead, the rule of `fee-switch-devnet.md`) and the three parameters; read the expected digest on a scratch node
first; `update-now`; the hand nodes and the seed with the same object; the digest sweep; the first paid segment
record (`igneum_getSegmentRecords`) and `igneum_getProvingStatus.v1.segmentsInWindow` after H.
6. **The mandatory rule** stays off: no switch exists for it yet; it gets one when the measured share is one.
## The decisions (Decided 5 October 2026, delegated: Josh, "I have no idea for most of this stuff so do a lot of research and deploy what is absolute best")
Each with its rule, its number and its evidence. The deploy is the DEVNET through 0.3.11 (not the public testnet).
### What the other networks do (read 5 October 2026, 20:05 to 20:15 UTC; every figure from the page named, else labelled approximate)
| Network | Unit proven | Deadline | What a miss costs | Who is paid what | Measured latency |
|---|---|---|---|---|---|
| Taiko Alethia (L2BEAT page, protocol v2.1.0 notes) | a batch of blocks | proving window 2 h, cooldown 2 h (v2.1.0, February 2025); the Shasta inbox targets a 4-h proof submission cadence | the proposer's liveness bond is credited back in full when the batch is proved inside the window, half when outside; mainnet currently sets minBond and livenessBond to 0 | the prover earns the proving fee; two of four proofs needed (SGX Geth, SGX Reth, SP1, RISC0, at least one ZK) | 100% ZK coverage of mainnet blocks reached December 2025 (blockchain.news); preconfirmations 2 s |
| Boundless (docs.boundless.network, proof lifecycle) | one request | the requester's timeout (example 3,600 s) and a lock timeout (example 2,700 s); a reverse Dutch auction ramps the price from the minimum to the maximum over a ramp-up (example 300 s) | the locked collateral (example 5 ZKC) is slashed and used to pay another prover who fulfils the request | the prover's fee = the bid minus the market fee | not stated on the page |
| Succinct Prover Network (docs.succinct.xyz, SPN architecture and quickstart) | one request | the requester's deadline (the quickstart example: 10 minutes, 50 PROVE staked to bid, 100 PROVE maximum fee) | part or all of the winning prover's collateral slashed "according to protocol rules" | a reverse auction: the lowest bidder is assigned | "real-time", no number on the page |
| Aztec (docs.aztec.network economics; L2BEAT; forum) | an epoch of 32 blocks (38 min 24 s), a proof may cover one checkpoint (1 min 12 s) up to one epoch; maximum proof window 1 h 16 min | the epoch is declared failed only when its submission window expires | an unproven epoch is reorged out (no reward); proposals under discussion remove bonds and pay every prover that delivers on time | 400 AZTEC a slot: 70% sequencers, 30% provers (120 AZTEC), provers' share by an activity score | the public testnet proved by community provers (zkcloud blog), no page number |
| zkSync Era, Linea, Scroll (eco.com comparisons) | a batch | none on chain (the operator proves) | none | the operator | proof latency about 30 min (zkSync Era), 75 min (Linea), 90 min (Scroll), approximate |
Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoever delivers the epoch proof, Taiko's fee to whoever proves the batch, the markets to the request's winner. Deadlines run from 10 minutes (Succinct's example) through 1 h (Boundless' example) to 2 h (Taiko) and 1 h 16 min (Aztec's maximum window); a miss forfeits the reward or part of a bond, and the slashed value goes to the prover who steps in (Boundless). Igneum has no bond on shards by decision (spec 7.2 item 4), so the forfeit here is the reward only.
### The decisions
| Decision | Decided | Rule and number | Evidence |
|---|---|---|---|
| `proving_v1_segment_blocks` (N) | **8** | Aggregation is a fixed cost per block, not per segment: 9.6 to 9.7 s for every chained block on a mining 5090, 7.9 s unchained (`chain-pc2-pv1c`), so N buys nothing in card time; it sets the record cadence and the forfeit. At N = 8 and 1 block/s a record every 8 s, a 1.27 MB proof gossiped every 8 s (159 KB/s per path, half of N = 4's 318 KB/s), and a missed segment forfeits 8 blocks' aggregator share (8 x 0.088 IGN at today's credit). The chain for 8 blocks cost 135.6 s cold on a mining card (66.8 s for 4), a fifth of T; pipelined per block it is 17 s after the last block. Aztec proves 32 blocks (38 min) as one; 8 blocks at 1 block/s is 8 s of chain, so the record lands well inside the minute the litepaper promises | bench-log "proving v1" chain rows; Aztec economics page |
| `proving_v1_unproven_daa` (T) | **600** DAA s (10 min) | T = p99 x 10: the measured block-to-carried-record latency of a shard record is p99 52 to 62 s (two 30-min windows), a cold chain of 8 adds 136 s and relay plus inclusion 10 to 40 s, about 240 s worst case; 600 leaves 2.5x on that and equals the 600-block record window of v0, so nothing is payable past it either way. Succinct's example deadline is the same 10 minutes; Boundless' example 1 h, Taiko 2 h, Aztec up to 1 h 16 min: Igneum's blocks are 1 s and its proofs seconds, so the shortest of the field. The forfeited aggregator share of an unproven segment STAYS IN THE POOL ESCROW (it is never paid, as an unproven shard's part today): no burn and no roll-over, the rule the pool already has, and the escrow is what later proofs are paid from | coverage rows; `chain-pc2-pv1c`; the table above |
| `proving_v1_aggregator_share_bps` | **1,000** (a tenth) | The aggregator's card time per block is 9.7 s on a mining card against 4 x 10.6 s of shard proofs at `B_p` (19% of the card time) and 2.5 s against 42.5 s with the card to itself (6%); on tonight's empty blocks it is half the card time. A tenth of every attested block's pool credit sits between the two full-block ratios, pays a role no other network pays separately (Aztec pays its 30% to whoever delivers the epoch; the markets pay the winner), and leaves the shard provers 90%, which the fast-time harness showed paid exactly (shardWei 90% of the credit). The pool's 20% emission share itself is unchanged (spec 2.5, 5.3) | `chain-pc2-pv1c`; bench-log 4 October 5090 rows; the harness |
| `proving_v1_activation_daa` (H) | **the devnet tip + 14,400 at publish** (4 h at 1 block/s), set by the 0.3.11 publisher in the same override object as `program_class_v3` | tonight's rule for consensus switches (the coordinator, 5 October 2026); the digest moves only once H is set, so the rolling update does not partition |
| Aggregator sortition | **none in v1**: the first valid record carried wins | design 5.3's VRF draw (O-7.3) with one or two aggregators on the devnet changes nothing; the segment grid and the deadline already bound the race; revisit when a second aggregator exists |
| Apple silicon default | **off** | the gate was "a shard under 60 s with the miner running": the M5 Max CPU took 41.3 and 55.4 s for EMPTY shards under tonight's load and 272 s for a 200-pgas shard on 4 October; a full shard at `S_p` was never under 60 s. Settings switches it on | bench-log "proving v1" CPU chain row; 4 October CPU rows |
| The prover profile per card and the 12 GB and 16 GB gates (Josh: "make sure we can prove on 12gb cards"; "is there any way we can make 12gb cards mine and prove?") | **measured on PC 2, the rows below** | the SP1 6.8.1 GPU server reads `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `SHARD_SIZE` and the `SP1_WORKER_NUM_*`/`BUFFER_SIZE` knobs from the environment it inherits (`sp1-core-executor-6.8.1/src/opts.rs`, `sp1-prover-6.8.1/src/worker/config.rs`); the app passes a profile per card (`provedefault.rs`) and the host forwards it | the sweep job `memsweep-pc2-pv1` and the miner-on run |
The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re-armed only FAULTED cards (`stop_miners("paused")` clears every slot's `restart_at`), so a healthy paused card stayed "off" at 0 MH/s until the app was relaunched: PC 2 at 21:25:11Z (the aggregation-cost job's pause and resume; `[ok] mining resumed` then `0.00 MH/s, waiting` for 20 minutes), the Mac that afternoon. Now every slot without a live worker is re-armed and its pack exported again before the start, and 90 s later `resume_check` logs `resume: <card> is not mining 90 s after resume (state ..., pid ...)` for every enabled card without a hash rate (`engine.rs`, three unit tests: the state machine, the 21:25:11Z case against the old rule, the check).
### A self-built CUDA server (the 12 GB path), before 0.3.12 (consequences C26)
If the prover-floor agent's rebuilt `sp1-gpu-server` (the Setup sizes cut, built on PC 2 under WSL2) proves a shard under 11 GB, it becomes a shipped artefact and needs its own row of rules before 0.3.12: it is built from a pinned SP1 source tag with `CUDA_ARCHS` covering sm_86, sm_89 and sm_120 (the 12 and 16 GB tiers are Ampere and Ada, not only the 5090's Blackwell; one card family per measured row), by the packaging path that builds the Windows payload (PC 1's build job for the Linux binary, the Mac signs the manifest as it does the DMG), lands in the DMG and the WSL2 package beside the host as `wsl2/bin/sp1-gpu-server` with its sha256 in `payload-inputs.json`, is named in `evidence.md` beside the prover rows ("prover built from SP1 <tag> at <sha>"), is rebuilt and re-measured at every SP1 upgrade, and ships only after `--mode verify-segment` and `--mode verify` on proofs it made show the pinned verifying keys unchanged (the server changes allocation, not the circuit; the ids `0x2b1a81cb...` and `0x474678f3...` must still verify them). The 12 GB claim itself waits for the on-order RTX 3060 to run that server on the same fixtures and recipe as the curve; until then the public line stays at 24 GB.
The root-socket class on PC 2, the two times: 20:00:56Z (my chain job's root run; the live prover failed with Connect(PermissionDenied) until the socket was gone) and 21:25:24Z (the aggregation-cost job's root run; the prover stayed dark through the 0.3.10 restart at 21:49:41Z until `socketfix-pc2-pv1` removed the root-owned `/tmp/sp1-cuda-0.sock` at 22:01:16Z; the next shard, block 89011, was proven at 22:02:13Z and paid, and every shard since). The permanent fix in the 0.3.11 app tree: every committed playbook that runs a prove mode as root carries `pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` (in `ci.yml`) fails a playbook without them, and the app's prover names the cause in its log line when the host reports PermissionDenied. The app itself cannot remove a socket another user owns, so a job written outside the tree must still follow the rule.
A prover job on a shared card runs as the app's user or cleans its socket (`pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at the start and the end; `tools/ci/prover-socket-check.sh`): the root-socket fault of 20:00Z, bench-log.
### The prover profiles: the tiers from the S_p curve (bench-log, "proving v1", the sweep, the miner-on pair and the curve)
The GPU server of SP1 6.8.1 sets the memory, not the shard: a floor of 13.9 GB for an empty shard, 20.4 GB for a full shard at the adopted v1 budget (30,000 pgas, 4.7 M cycles), 28.3 GB for the prototype shard (6.75 M pgas, 60 M cycles); the miner adds 1.7 GB when it shares the card; no environment knob moves the floor and the server has no options of its own; the witness is 5 to 22 KB a shard and never binds.
| Card | Alone | Beside the miner | Default (`provedefault.rs`) |
|---|---|---|---|
| 32 GB (RTX 5090) | the prototype shard, 28.3 GB, 10.8 s; the v1 shard 20.4 GB, 4.3 s | the prototype shard 30.1 GB, 33 s; the v1 shard 22.2 GB, 13.2 s | on, mine and prove, today |
| 24 GB (RTX 4090, 3090) | the v1 shard 20.4 GB; the prototype shard does NOT fit (28.3 GB) | the v1 shard 22.2 GB measured on the 5090's allocation (2.3 GB spare on a 24 GB card; approximate for the card itself) | on, mine and prove, with the line "until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" |
| 16 GB (RTX 5080, 4080) | an empty shard only (13.9 GB) | nothing (15.7 GB for an empty shard, no room for the display) | off, with the line |
| 12 GB (RTX 3060, 4070) | nothing: the floor is 13.9 GB, and the shipped server refuses the card outright | nothing | off; Josh's "make sure we can prove on 12 GB cards" is OPEN and in work: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) |
| under 12 GB | nothing | nothing | off, mine only |
| AMD-only and Apple machines | nothing on the GPU: no zkVM proves on an AMD GPU today (`docs/analysis/amd-proving.md`, branch amd-prove); the CPU prover is about 5 minutes a shard at a 30 GB RSS whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1") | | off, "mines and does not prove"; the only non-NVIDIA path with a shipped backend is RISC Zero's Metal prover behind the `ProofSystem` seam (a second guest and pinned id, a verifier for both formats, no shared aggregation): an open item, not 0.3.11 |
The three profile numbers the coordinator asked for, as measured: under 9.0 GB does not exist on this build (floor 13.9); under 15.0 GB mine-and-prove does not exist for any full shard (the v1 shard alone is 20.4); the full profile is the 32 GB card. The fleet table's "proving-only" rows therefore read 24 GB cards at the v1 budget and 32 GB cards at the prototype budget. Shards per block at the v1 budget: 1 on tonight's empty chain, 2 to 4 on blocks with transactions (`B_p` 120,000 = 4 x `S_p`); the aggregation count is one per block whatever the shard count (the chained recursion), so the aggregation-cost agent's target is per block.
The aggregation-cost agent's first rows (branch agg-cost, 5 October 2026 night, the same four live blocks on PC 2): with the miners paused an empty shard proves in 1.9 to 2.2 s and an aggregation in 1.7 to 2.2 s with the card 15.8% busy; mining, 7.4 to 7.8 s and 7.9 to 9.8 s at 93.9% busy, so the miner's kernels take the card and the prover runs 3.6x (shards) to 4.5x (aggregations) slower beside them; its batch-size curve is still open. That puts a proving-only card at about 4 s per empty block (one shard and one aggregation), 4 cards for an empty-block chain at 1 block/s, against 18 mining cards.
The re-plans of block 344 at 2.25 M and 4.5 M pgas peak at 28.3 to 28.4 GB alone (the server's buffers step up between 4.7 M and 20 M cycles and are flat to 60 M), so no shard size between the v1 budget and the prototype one changes a tier; with the miner the adopted shard proves 3.1x slower (13.2 s against 4.2 s) and the chained aggregation 9.7 s against 2.5 s: a mining 24 GB card delivers one adopted-size shard plus one aggregation in about 23 s, inside T by 25x.

126
docs/plans/read-width.md Normal file
View file

@ -0,0 +1,126 @@
# Read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, and a written scratch (gate 1 experiment)
5 October 2026. Branch `readwidth` (worktree `../igneum-wt-readwidth`), commits 019b014 and b970dda plus the measurement commit. Nothing here changes consensus, the live generator, the pinned vectors or a shipped binary: every class sits behind `--class` in `igneum-pow` and the default class is generator version 2 byte for byte (`igneum-pow/tests/packs.rs` still compares the four pinned packs against the emitters). Numbers and a recommendation; the decision is Josh's.
## 1. The question
The bench-log entry "the 9070 XT on the eGPU" (5 October 2026) found the hash bound by dependent random 4-byte reads over the 1 GiB dataset, 128 per hash: the RX 9070 XT finishes 2.4 to 2.7 G such reads a second (18 MH/s), the RTX 5090 16 to 18 G (127 MH/s), the M5 Max 3.45 G (23 to 28 MH/s). AMD fetches a 64-byte line per 4-byte read, so 94 percent of its memory traffic is unused; NVIDIA fetches a 32-byte sector and its 96 MB L2 catches a share. Josh's question: would wider reads keep the chip-resistance property (latency-bound, random access) while closing the vendor gap? Two additions from the coordinator: a per-load width drawn from an era-fixed mix so no chip is built for one width, and a written per-warp scratch so part of the memory work cannot be mirrored into read-only SRAM.
## 2. What was built (all behind the flag)
| Class (`--class`) | Loads per hash | What a load does | Dataset bytes per hash |
|---|---|---|---|
| `v2` (= `w4`, the lottery hash) | 128 | `dst ^= dataset[src & MASK]`, one 4-byte word | 512 |
| `w16` | 128 | the 16-byte-aligned group of 4 words at `src & MASK`, every word folded into `dst` | 2,048 |
| `w64` | 128 | the 64-byte-aligned item (16 words), every word folded | 8,192 |
| `w64x4` | 32 (4 load slots) | as `w64`; the same bytes per hash as 512 loads of 4 bytes | 2,048 |
| `mix50-35-15` | 128 | per load, width 4, 16 or 64 bytes drawn from the program stream with probabilities 50/35/15 | 1,664 to 3,680 over the six programs measured (expected 2,202) |
| `mix25-50-25` | 128 | the same with 25/50/25 | 2,240 to 5,024 (expected 3,200) |
| `scr<k>k<kb>` | 128 memory operations | `k` of the 16 slots are scratch read-modify-writes into a `kb` KiB per-warp scratch (16-byte slots, lane-major); the other `16 - k` are 4-byte loads | 4 x (16 - k) x 8 reads plus 16 B read and 16 B written per scratch op |
The fold. A load of W words reads the W-word-aligned address `b = (src AND MASK) AND NOT (W - 1)` and sets `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) XOR w[j]; dst = x` (`verify::fold_words`, mirrored in the three kernel dialects). The rotate-multiply between the words makes the fold state-dependent: two different lines give two different maps of `dst` (the multiply by an odd constant is not xor-linear), so no function of the line alone can stand in for it, and a dataset of pre-folded lines cannot replace the dataset. The dependent chain is unchanged: the next load's address comes from a register that the fold wrote. Width 1 is the lottery hash's xor of one word, so `w4` is the pinned program `bcc1248b10cc90f2` exactly.
The per-load draw. Every class other than `v2` takes one extra draw per instruction (`below(100)`, the width roll, consumed on every slot so the stream stays uniform) after the nine draws of spec 01 section 1.4.3; on a load slot the width is the first entry of the mix whose cumulative weight exceeds the roll. The program id of a class is `FNV-1a-64("igneum-program-rw/" || 2 || seed words || attempt || mix[3] || slots [|| "scratch/" k kb])`, so no class program can pass for a version 2 program. The acceptance rule of 1.4.6 runs unchanged on the aligned addresses (lane-constant sites and the distinct-address bound, scaled to the dataset loads per hash).
The scratch (variant 5, measurement only). The kernel runs N persistent warps (one per block or work-group of 32); warp `w` owns scratch `w` and runs units `w, w + N, ...` of the launch. A scratch op reads the lane's 16-byte slot `src AND (slots - 1)`: three data words behind a per-unit tag; a slot whose tag is not this unit's reads as its fill `splitmix32(((base + lane) XOR seed[j]) + slot x 0x9e3779b1 + (j + 1) x 0x85ebca77)`, the three words are folded into `dst` as above, and the slot is rewritten `(tag, x XOR w1, rotl(x, 7) XOR w2, x + w0)`. The CPU verifier holds the touched slots of one unit (at most 32 x k x 8) and nothing else. The scratch is per lane (a 32 KiB warp scratch is 64 slots per lane, 128 KiB is 256), so two lanes never race on a slot and the result is a function of (program, day, unit) alone; the GPU's tags make the lazy fill exact as long as a tag is not reused within the arena's history (the salt advances per unit; it wraps after 2^32 units, a measurement caveat, not a design).
## 3. Method
| Step | Command (every figure in the bench log carries its command) |
|---|---|
| Packs | `igneum-pow export --seed <s> --class <c> --out proto-cuda/packs-readwidth/<name>` (23 packs; the six mix seeds per mix are `igneum-readwidth/A/<k>` and `/B/<k>` with attempt 0 accepted, `A/4` skipped: rejected at attempt 0) |
| CPU verifier | `igneum-pow bench --seed igneum-genesis --class <c> --warps 50` (M5 Max, one core; load average 4 to 9 from other agents' builds during the run) |
| Metal | `proto-metal/packbench --pack <dir> --batches 5 --batch-log2 24 --group 256 [--warps N]` (new harness: runs the pack's own text; vectors, cache FNV, dataset words, 2^24 fingerprint, MH/s by GPU time) under the measure lock |
| Apple OpenCL | `proto-opencl/igneum-bench-cl-rw --bench-pack --pack <dir> --batches 5 --batch-log2 24` and `--memprobe` (Apple's OpenCL, a correctness check and an approximate rate) |
| Emulators | `proto-cuda/emu/emu.sh ../packs-readwidth/<p> --batch-log2 13 --batches 1 --block-warps 2` (the CUDA text as C++); `proto-opencl/emu/emu.sh ../packs-readwidth/<p> 0 32 --sg 32` and `1 64 --sg 64` (the OpenCL text, wave32 and wave64) |
| RTX 5090 | job `run-readwidth-5090-20261005` on PC 2 (`relay/playbooks/readwidth-5090.ps1`): the NVIDIA card switched off in the app through `POST app.url/api/cards` and restored after; `igneum-worker-cuda --memprobe`, then `--bench --pack <dir> --batches 5 --batch-log2 24` per pack (NVRTC, the pack's own text, vectors through the bound kernel, 2^24 fingerprint) |
| RX 9070 XT | job `run-readwidth-9070-20261005` on PC 1 (`relay/playbooks/readwidth-9070.ps1`): only the gfx1201 card switched off; `igneum-worker-opencl --device D --memprobe`, then `--bench-pack --pack <dir> --batches 5 --batch-log2 24` per pack |
Latency-bound share = measured MH/s x loads per hash / the card's dependent-read ceiling for that width from its own probe at 1024 MiB (for a mix, the harmonic combination of the widths' ceilings weighted by the program's width counts). A share near 1 means the hash runs at the card's random-access limit, the property the design wants; a share well under 1 means something else bounds it (bandwidth, ALU, occupancy).
## 4. Results (full tables with commands in `docs/bench-log.md`, "read width of the lottery hash")
Probe ceilings at 1024 MiB (G dependent reads/s): RTX 5090 4 B 17.5, 16 B 18.0, 64 B 9.1 (584 GB/s), stream 1,579 GB/s; RX 9070 XT 4 B 2.42, 16 B 2.43, 64 B 2.47 (158 GB/s), stream 636; M5 Max (Apple OpenCL, approximate) 3.50 / 3.51 / 3.51, stream 522.
| Class | dataset B/hash | RTX 5090 MH/s (latency-bound share) | RX 9070 XT (share) | M5 Max Metal (share) | 5090 / 9070 | DRAM bytes moved per hash, NVIDIA 32 B sector / AMD 64 B line | CPU verify ms per unit |
|---|---|---|---|---|---|---|---|
| v2 = w4 (today) | 512 | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x | 4,096 / 8,192 | 0.604 |
| w16 | 2,048 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x | 4,096 / 8,192 | 0.610 |
| w64 | 8,192 | 71.9 (0.58) | 17.59 (0.78) | 28.27 (1.03) | 4.1x | 8,192 / 8,192 | 0.630 |
| w64x4 (32 loads) | 2,048 | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x | 2,048 / 2,048 | 0.160 |
| mix50-35-15 (6 programs, min / median / max) | 1,664 to 3,680 | 99.5 / 114.2 / 121.0, spread 18.8% | 17.45 / 18.76 / 18.83, 7.4% | 25.36 / 27.26 / 28.43, 11.3% | 6.1x | 5,939 / 8,192 expected | 0.620 |
| mix25-50-25 (6 programs) | 2,240 to 5,024 | 95.9 / 107.3 / 119.8, 22.3% | 17.84 / 18.45 / 18.85, 5.5% | 23.21 / 24.68 / 25.21, 8.1% | 5.8x | 7,168 / 8,192 expected | 0.614 |
### 4.1 Per watt and per pound (consequences review C11)
The runs carried no power sampling; the watts are the telemetry entry's (`docs/bench-log.md`, opencl-rdna4-telemetry, 5 October 2026: the RTX 5090 at 307.6 W under its 450 W cap for 122.3 MH/s, the RX 9070 XT at 199 W of its 304 W rating for about 17.8 MH/s, both on v2 with the shader clock at its top and the die waiting on memory), held constant across classes because every class is memory-bound on both cards (approximate: a class that moves more bytes per hash draws somewhat more at the memory controller, unmeasured). The Mac's GPU power is not measurable without root (`powermetrics`) and is taken as about 50 W (approximate, from memory). Prices are UK list, approximate, from memory.
| Class | RTX 5090 MH/W (at 307.6 W) | RX 9070 XT MH/W (at 199 W) | 5090 / 9070 per watt | M5 Max MH/W (at about 50 W GPU, approximate) | 5090 MH per pound (at about 1,900, approximate) | 9070 XT MH per pound (at about 570, approximate) |
|---|---|---|---|---|---|---|
| v2 (w4) | 0.442 | 0.091 | 4.9x | 0.55 | 0.072 | 0.032 |
| w16 | 0.454 | 0.090 | 5.1x | 0.57 | 0.074 | 0.031 |
| w64 | 0.234 | 0.088 | 2.6x | 0.57 | 0.038 | 0.031 |
| w64x4 | 0.895 | 0.378 | 2.4x | 2.19 | 0.145 | 0.132 |
| mix50-35-15 (median) | 0.371 | 0.094 | 3.9x | 0.55 | 0.060 | 0.033 |
| mix25-50-25 (median) | 0.349 | 0.093 | 3.8x | 0.49 | 0.056 | 0.032 |
| scr8k32 | 0.397 | 0.071 | 5.6x | 0.98 | 0.064 | 0.025 |
| scr2k32 | 0.372 | 0.074 | 5.1x | 0.52 | 0.060 | 0.026 |
Reading: whatever width is chosen, an AMD home miner keeps about a seventh of a 5090's rate and pays about 4.5x the electricity per hash, because every width costs the 9070 XT the same 2.4 G line fetches a second; per pound of card the 5090 is 2.2x the 9070 XT at v2 and w16 (0.072 against 0.032 MH/s per pound) and 4.9x per watt; only w64x4 narrows the per-pound gap (0.145 against 0.132), and that class fails the width rule. The consequence for the decision (D6, Josh's): AMD's line width is not a read-width question at all; it is the card's random-access rate, and the levers that act on it (the 64 MB Infinity Cache against the dataset size, the memory path) are v3-or-3.0 questions outside this experiment.
Scratch, variant 5 (N persistent warps; GPU cost against the persistent control scr0k32; working set = 1 GiB + 256 MiB + 128 MiB output + N x size):
| Class | RMW share | dataset B/hash | scratch B/hash read + written | RTX 5090 MH/s, 2,048 warps of 4,080 resident (vs control, share) | M5 Max Metal (vs control) | RX 9070 XT, 4,096 warps | working set 5090 / 9070 / Mac |
|---|---|---|---|---|---|---|---|
| scr0k32 | 0 | 512 | 0 | 139.1 (control, 0.98) | 28.25 (control) | 17.88 (control, 0.86) | 1.4 GiB / 1.5 GiB / 1.5 GiB |
| scr2k32 | 12.5% | 448 | 256 + 256 | 114.4 (-18%, 0.80) | 26.14 (-7%) | 14.65 (-18%) | same |
| scr4k32 | 25% | 384 | 512 + 512 | 109.8 (-21%, 0.76) | 31.74 (+12%) | 14.00 (-22%) | same |
| scr8k32 | 50% | 256 | 1,024 + 1,024 | 122.1 (-12%, 0.82) | 49.08 (+74%) | 14.17 (-21%) | same |
| scr2k128 | 12.5% | 448 | 256 + 256 | 110.1 (-21%, 0.77) | 26.24 (-7%) | 14.07 (-21%) | 1.6 GiB / 1.9 GiB / 1.9 GiB |
| scr4k128 | 25% | 384 | 512 + 512 | 98.0 (-30%, 0.68) | 28.08 (-1%) | 13.14 (-27%) | same |
| scr8k128 | 50% | 256 | 1,024 + 1,024 | 72.8 (-48%, 0.49) | 35.44 (+25%) | 12.03 (-33%) | same |
Resident warps and the cap: the 5090 holds 4,080 warps at one warp per block (24 blocks per SM x 170 SMs; 8,160 at 8 warps per block), so 128 KiB each is 510 MiB and the whole working set 1.9 GiB; a 1 MB scratch would have been 4.0 GiB at this geometry and 10.6 GiB at the 64-warp-per-SM figure, which is why the cap moved the size to the tens of kilobytes. The occupancy query returned 24 blocks per SM before and after the arena allocation: the allocation did not change it. The 9070 XT's OpenCL runtime has no occupancy query; 4,096 persistent warps were launched (64 per compute unit over 64 CUs, approximate) and the arena is 128 MiB at 32 KiB, 512 MiB at 128 KiB. The Mac's residency is not reported; 2,048 to 16,384 warps were swept and the best row kept.
Chip model, re-run with the measured widths (the M16 arithmetic of `docs/analysis/m16-recompute-attacker-2026-10-05.md`; the on-die-cache recompute chip's row per scratch variant is the ca2-soundness branch's, as agreed with the Counter ASIC 2.0 coordinator):
| Class | what a chip with its own DRAM controller gains over the GPU's memory system | what a chip with on-die SRAM gains |
|---|---|---|
| v2 | the GPU fetches 8 to 16x the bytes it uses (AMD 64 B, NVIDIA 32 B per 4 B); a chip fetching 32 B bursts moves 4,096 B per hash, the 5090's figure, so nothing over NVIDIA and 2x over AMD in traffic, none in latency (the chain is 128 dependent DRAM latencies on either) | the recompute attacker of M16: 150,000 integer ops per hash against the 256 MiB cache; 2.4x at equal silicon before a fixed-function factor (unchanged by the width) |
| w16 | the same: 4,096 / 8,192 bytes moved, 2,048 used; traffic efficiency 50 percent on NVIDIA, 25 on AMD | unchanged: the fold uses every byte, so the chip recomputes 128 items per hash exactly as before; the SRAM mirror of the read-only dataset (1 GiB) stays out of reach |
| w64 | every byte moved is used on both vendors (8,192 moved, 8,192 used); the 5090 is bandwidth-bound at 589 GB/s, so a chip with HBM3 class bandwidth (several TB/s, approximate) is bandwidth-advantaged: the Ethash shape | unchanged in op count; but the chain of 128 loads now moves 8 KB, so a chip's advantage shifts from latency to bandwidth per dollar, which is the wrong direction for the design's 2x target |
| w64x4 | 2,048 moved and used; 32 latencies per hash; every card 4x faster; a bandwidth-rich chip gains as above | 32 items per hash: the recompute attacker's op count falls 4x (37,500 per hash), so the M16 gain rises 4x: fails the 2x target by arithmetic |
| mixes | between v2 and w64 per program; the chip cannot be built for one width, but the GPU pays the 64-byte hours (the 5090 loses up to 27 percent in a heavy hour) | as v2 per item; the recompute attacker is indifferent to the width |
| scratch | a chip must provide writable memory for N units in flight: 32 KiB x N at the GPU's geometry (128 MiB at 4,080), against the 256 MiB read-only cache it could mirror into SRAM (54 to 83 mm^2 at a leading node, the coordinator's figure, approximate); but a unit touches at most k x 8 x 32 slots (4 KiB at 50 percent), the fill is a function and the tags are per unit, so a chip need only hold the touched set per unit in flight (the soundness caveat below) | the dataset reads replaced by scratch ops are reads the chip no longer has to serve from the 1 GiB; at 50 percent the recompute attacker computes 64 items instead of 128 |
Soundness (variant 5, measurement only, as instructed; the chip row is the ca2-soundness branch's, a465881: the on-die-cache recompute chip's gain is 2.4x at 0, 12.5, 25 and 50 percent, replaced or added, 32 or 128 KB, so the scratch does not move it): the per-unit scratch starts from a fill that any implementation can compute, and a unit writes at most `k x 8` slots per lane; an implementation that keeps only the touched slots of each unit in flight (the CPU verifier does exactly this) needs 16 B x touched slots, not the nominal arena, so the "real memory a chip must provide" is bounded by units in flight x touched slots, not by N x 32 KiB. The variant forces memory that is written, which SRAM can hold as well as DRAM; it does not force memory that is large. A written region that outlives the unit (state carried across units) would, and the CPU verifier could not replay it. This is the finding, not a recommendation.
## 5. Recommendation (the decision is Josh's)
Josh's rules, as passed by the coordinator: width = the widest read that keeps every card latency-bound (achieved within 90 percent of the probe ceiling at that width) with margin on the 5090 (bytes per hash x rate under a third of the 1,579 GB/s stream); the mix is in only if the six-program spread is under 5 percent per card; the scratch share is the smallest at which the chip model's gain falls under 1.5x at the lowest GPU cost within the 6 GB cap.
| Variant | Verdict under the rules | Numbers |
|---|---|---|
| w16 (16-byte loads, 128 per hash) | PASSES the rules: shares 0.90 / 0.84 / 1.03 (the 9070 XT's 0.84 equals its v2 share of 0.87 within noise: the card is at its ceiling in both), 286 GB/s on the 5090 = 18 percent of the stream. It does NOT close the vendor gap (7.8x against 7.5x), because the memory systems already move a sector or a line per load; it changes what the fold consumes, nothing the DRAM does | the only width row that passes; a no-cost change in rate (+2.7 percent 5090, -1.4 percent 9070 XT, +1.9 percent M5 Max) |
| w64 | FAILS: 5090 share 0.58, 37 percent of the stream; closes the gap to 4.1x only by making the 5090 bandwidth-bound | |
| w64x4 | FAILS: shares 0.56 / 0.84 / 1.00, the recompute gain rises 4x | |
| mix 50/35/15 and 25/50/25 | OUT: spreads 18.8 and 22.3 percent on the 5090, 7.4 and 5.5 on the 9070 XT, 11.3 and 8.1 on the M5 Max, all over 5 percent; a chip is not built for a width anyway (see the model: the width does not change the recompute attacker) | |
| scratch | OUT: every share costs the 5090 12 to 48 percent and the 9070 XT 18 to 33 percent, and raises the M5 Max's rate (the arena is cached there); the soundness branch's chip row (ca2-soundness a465881, the on-die-cache recompute chip) stays at 2.4x at every share, 32 or 128 KB, because the verifier resets the scratch per unit and the live state is the hash's own read-modify-writes, which a chip keeps in 80 to 320 B per lane; under the rule the share is 0. The rows stay as the measurement that decided it | |
Recommendation: keep 128 loads per hash and 4 bytes per load (v2) for the devnet; if a width change is wanted for the fold's sake (every byte of the sector consumed, which removes the "94 percent waste" statement from the AMD entry without changing what the card does), w16 is the one that passes every rule and costs nothing measurable, and it is the only width worth a vector re-cut. The AMD gap is a random-access gap (2.4 G against 17.5 G dependent reads per second at 1 GiB on the cards we own), and no read width closes it without turning the 5090 bandwidth-bound; the levers that act on the gap are the ones outside this experiment (the AMD card's memory path, and the dataset size against the 5090's 96 MB L2 share, which the probe's 64 MiB rows show at 9 G reads/s against 2.4 at 1 GiB). The per-load mix is out on stability; the scratch is out on GPU cost and on the soundness caveat.
## 6. What w16 would change if adopted (not done; the decision is Josh's)
| Where | Change |
|---|---|
| `docs/spec/01-lottery-hash.md` 1.4.1 | `load`: `dst = fold(dst, dataset[b .. b + 4))`, `b = (src AND MASK) AND NOT 3`, with the fold written out; 1.4.3 unchanged (no width draw for a fixed width); 1.4.6 unchanged (the aligned address is the address the rule sees) |
| 1.5 | "A load reads one 4-byte word" becomes 16 bytes aligned; the single-form text-search rule of 1.14 item 2 becomes the wide form; the item size (64 B) and `dataset[w] = item(w >> 4)[w AND 15]` unchanged |
| 1.11 | unchanged in count (4,096 items per unit; the verifier derives the same items) |
| 1.15, 1.17 | every vector re-cut (new program ids: the class enters the id or the generator version steps to 3); the four pinned packs replaced; the conformance fuzz re-run on Metal, CUDA and OpenCL (this branch's 23 packs and the three emulators are the template) |
| Litepaper, Mining ("random reads over a multi-gigabyte dataset") and the vs-RandomX "128 dataset addresses" rows | "128 reads of 16 bytes"; `site/bench.html` sector arithmetic (32 B per 4 B) becomes 32 B per 16 B |
| Workers | no host change: the kernel text carries the loads; `proto-cuda/host.cu`'s static mask check (`TESTS.md` section 5) learns the wide form |
| Cost on the 5090 | none measured (+2.7 percent); on the 9070 XT -1.4 percent; verifier +1 percent |
## 7. Files
`igneum-pow/src/{generator,verify,accept,emit,memhard,main}.rs` (the classes, behind `--class`), `proto-cuda/packs-readwidth/` (23 packs), `proto-metal/packbench.swift` (Metal from a pack's files), `proto-opencl/host.c` (`--bench-pack`, `--warps`, the 16-byte probe row, the scratch arguments), `proto-cuda/nvrtc/worker.cpp` (`--bench`, `--memprobe`, the scratch arena), `proto-cuda/nvrtc/packfile.h` (class fields; string-seed packs), `proto-cuda/emu/cuda_runtime.h` and `proto-opencl/emu/{emu_opencl.h,emu_main.cpp}` (vector types, the persistent launch), `relay/playbooks/readwidth-*.ps1` (the PC jobs: the card under test off in the app and restored, never the other card).

View file

@ -0,0 +1,232 @@
# Igneum Miner 0.3.11: program class v3 (Counter ASIC 2.0) and proving v1 on the devnet, 5 October 2026
Release engineer, from 22:39 UTC, on the coordinator's instruction under Josh's delegation ("Counter ASIC 2.0 fully deployed",
"deploy what is absolute best" for proving v1). Worktree `/Users/joshm/Projects/igneum-wt-ship0311`, branch `release-0.3.11`,
assembled by the Counter ASIC coordinator from master b38f3de (the 0.3.10 merge) and taken over at its merge tip b968ee0 so there
is one ship, not two. Fork worktree `vendor/igneum-node-0311` UNDER the release tree (the node links `../../../../igneum-pow`, the
release tree's crate), branch `release-0.3.11-node` at 89dfcb95 (on 21d4c73c = 0.3.10's node, with proving-v1 ece42979 and the
digest re-pin). The 0.3.10 recipe (`release-0.3.10.md`) throughout; every Mac build under the main checkout's lock; every PC
job and publish from this worktree's tools (the signed envelope, the per-job zip names). Times are UTC.
## 1. What 0.3.11 carries
| Change | Where | State |
|---|---|---|
| Program class v3 (Counter ASIC 2.0): the era draw, the cache growth rule, the mixer x8, behind `program_class_v3_activation_daa` (keys on the epoch: `program_class_v3_first_epoch`); the workers carry the class, era and attempt rules on the serve protocol and refuse a pack of the wrong class or era | main `ca2-v3` fa3c932 (code 49c7e78; `igneum-pow`, the workers, the fast-time scripts, the plans); fork `ca2-v3-node` 89dfcb95 | merged (c9b0b2c) |
| Counter ASIC 2.0 docs, spec, site, evidence, the rollout plan and its gates (G1, G2, G3, G4, G4b, G6 green; G5 = the one-commit workers of this cut) | main `ca2-coord` 076c0ab then 57844e9 (C34) | merged (18605f4, 5cedcd4) |
| Proving v1 (spec 7.8): the aggregated segment record, the chain rule and the unproven rule behind `proving_v1_activation_daa`, with `proving_v1_segment_blocks` 8, `proving_v1_unproven_daa` 600, `proving_v1_aggregator_share_bps` 1000; the app's prover loop (the CPU path refused under 32 GB with the reason, the root-socket cleanup, the PermissionDenied line naming the cause); the resume fix (every stopped card re-armed, its pack re-exported, checked 90 s later); every worker gets `--prepare-packs` in the platform's path form (the Mac's Metal worker takes a class v3 day from the prepared pack) | main `proving-v1` 22c2363 (its agent: the final code tip; c36dfea after it is docs only and waits); fork proving-v1 ece42979 inside 89dfcb95 | merged (5cbb796) |
| CI: `bash-body-check.sh` (inline bash bodies in PowerShell job scripts parse), `kit-path-check.sh` (a run job tests its fetched kit before use, C32), `prover-socket-check.sh` (every root prover playbook unlinks the GPU server's socket) | main `bash-body-check` e3bd761 | merged (fe1ecdb; ci.yml keeps the signer-pipe step, both new steps and proving-v1's socket step) |
| The consequences ledger, the proving-methods analysis, the ASIC-resistance history | `consequences` 99fd988, `proving-methods` e7e0db7, `asic-history` 9e4af7f | merged (5dffb1b, 0f5bfc3, b968ee0) |
| The pinned proving guests | unchanged (no prover drain) | |
Changelog line (the coordinator's words): "Igneum Miner 0.3.11: program class v3 (the era draw, the cache growth rule, the mixer x8) from epoch
N4/3600 and proving v1 from DAA N5; the resume fix; the Metal worker takes v3 from a prepared pack".
## 2. The branch
| Commit | What |
|---|---|
| b968ee0 | the Counter ASIC coordinator's merge tip (above), taken over at 22:40Z; its checks on the Mac: igneum-pow 53 + 4 + 19 + 7, the app 113 + 27 + 8, 0 failed |
| 21173c4 | `Igneum Miner 0.3.11: the six version files` (`--check`: 0.3.11 in all 6) |
| cc72f4a | CI on the merged tree, three findings fixed: the identity check's hostname pattern `MacBook` matched prose in `card-lifetime-2026-10-05.md` and `proving-methods.md` (reworded "Apple laptop"); the kit-path check flagged `tools/proving-v1/pc2-{memory-miner-on,memory-sweep,sp-curve}.ps1` for a bare `jobs\` literal in `WslPath (Join-Path ...)` (now `Test-Path` on the kit root, then `WslPath $kitFile`); the socket check flagged `tools/amd-prove/pc1-cpu-prove.ps1` and its `-sp` sibling, which the addendum said were allow-listed and were not (allowed, the CPU path starts no GPU server). The other agent's resolution had kept both ci.yml sides and bash-body-check's 24-line socket check; one slip of mine (a `git show :3:` redirect after that resolution had already committed) truncated that script to 0 lines in the working tree for a minute and was restored from HEAD |
| 23bc2b2 | `packaging/mac/packaged-config.sh`: the nine-field object (section 4) in the packaged line, `--test` passes (C34: a fresh install must start on the fleet's digest) |
| a94416e | the plan draft committed to the tree before the push (the reviewer's C37) |
| after a94416e | C38 (the reviewer through the Counter ASIC coordinator): the evidence rows, the litepaper's chip bullet, `chip-model-v3.md` and the rollout plan cite files on four Counter ASIC branches that never reached the tree. Taken as docs plus standalone sources under one gate: the diff against 23bc2b2 over every input a built artefact reads (`igneum-pow/src`, `app/`, `proto-cuda/nvrtc/{packfile.h,worker.cpp,cuda_api.h,build-windows.sh}`, `proto-cuda/{host.cu,build.bat,windows-app}`, `proto-opencl/{host.c,cl_dynamic.h,build.*}`, `proto-metal`, `packaging`, `proving`, `vendor`, `.github`) must stay empty apart from files no script compiles or copies. Merged: `ca2-analysis` ee42d7c (`sram-mirror.md`, `int8-matrix-family.md`, the dot4 probe sources: `proto-metal/dot4-probe.swift` is standalone, the DMG script compiles `main.swift` alone), `ca2-epoch` e95e8b5 (`epoch-length.md`), `prover-floor` cfe3d80 (`prover-floor.md`, `tools/prover-floor/` scripts, a playbook, `proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, which nothing reads: the next cut's packaging row); `docs/bench-log.md` conflicted each time (append-only: both sides kept). `ca2-soundness` a465881 conflicted in `igneum-pow/tests/scratch.rs` (add/add) and `proto-metal/packbench.swift`, code the mixer merge already carries, so its merge was aborted and `docs/analysis/scratch-soundness.md` taken alone. So G5 holds: the workers, the Metal worker and the DMG built at 23bc2b2 correspond to the tip's built inputs |
Every check at cc72f4a: identity 0 hits over 220 files, copied-sources, pinned-guests, signer-pipe, bash-body 15 bodies in 28 files, kit-path 14 of
14 kits checked, prover-socket, no-conflict-markers, workflow shell 0 findings over 35 .ps1, relay tests 17, UI tests 23.
## 3. Builds and tests (the 0.3.10 recipe)
| What | Command | Result |
|---|---|---|
| The fork's Mac node, 89dfcb95 | `CARGO_TARGET_DIR=vendor/igneum-node/target-0311 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` from `vendor/igneum-node-0311` (under the release tree), under the lock; the target dir cloned by APFS from `target-0310` | 22:46:16 to 22:49:5xZ (3 min 19 s, incremental): igneumd bd7f043c453f4b3e9575e678912b71115227b09789d5ec0f367b8278f031d043 (41,386,016, `89dfcb95` in its strings); copied into the fork worktree's `target-integration/release/` |
| The seed's Linux node (glibc 2.36 target, zig) | `NODE_SRC=<abs fork> TARGET_DIR=<abs> OUT_DIR=<abs> infra/cross/build-linux.sh` (the 0.3.10 fix: absolute paths; `cargo clean -p kaspa-build-info` first), under the lock | 22:46:35 to 22:49:59Z (3 min 20 s): igneumd 63cf490d42483d3aa4e525eba9b4e4b68cae9090c32f409e745858defc381c52 (47,913,832, `GLIBC_2.34` at most, `89dfcb95` in its strings), igneum-miner a136d622... (9,861,280) |
| The two Windows workers (G5: one commit) | `proto-cuda/nvrtc/build-windows.sh` under the lock, tree 23bc2b2 | 22:46:43 to 22:46:50Z: igneum-worker-cuda.exe 2b3b8c92885442179f6bf2907c6f3eb453dc4a19908d90fd05981a09b7c2674c (1,536,512), igneum-worker-opencl.exe edc4a75da3b93d814caa69fd635010780d63d5b622ec24c3741d433c584f91e3 (478,208); both carry the resource block; both differ from 0.3.10's pair (85cc357b..., afa73a32...): the class, era and attempt rules are in them |
| The app | `cargo build --release` in `app/igneum-app` (target cloned from the 0.3.10 worktree), then `cargo test --release -p igneum-app`, under the lock | 22:47Z: `igneum-app 0.3.11`; tests ok 113 (lib) + 27 (ota-sign) + 8 (prove-verify), 0 failed |
| `igneum-pow` | `cargo test --release` in `igneum-pow`, under the lock | 22:47:33Z: ok 53 + 4 + 19 + 7, 0 failed (the class v3 vectors, the era draw, the mixer x8, the scratch soundness) |
| The prover host and export (the pin unchanged) | `cargo build --release -p igneum-prove-export -p igneum-prove-host` in `proving/igneum-prove` (the worktree needed the vendor links: `vendor/igneum-node-exec` and 45 others symlinked to the main checkout's, beside the real `igneum-node-0311` worktree) | 22:48Z: `--mode id` shard `0x2b1a81cb...`, aggregator `0x474678f3...` (the 0.3.9 pin: no prover drain); the nine real fixtures `--mode native` all ok |
| PC 1 build job (the node and the app, Linux and Windows) | `IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release node tools/build-job.mjs run --node vendor/igneum-node-0311 --target ae432dc7 --targets linux,windows --no-tests` from this worktree, its own zip `build-inputs-20261005224625-29789.zip` | job `build-20261005-224654`, published 22:46:54Z (PC 1 given by the Counter ASIC coordinator at 22:4xZ: Ember's collect job closed, the AMD sweep off tonight). (pending) |
| PC 2 combined job | `IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release node tools/build-job.mjs run --node vendor/igneum-node-0311 --target 1ccfe586 --targets linux,windows --node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows" --app-tests igneum-app` from this worktree (its own zip `build-inputs-20261005230716-50058.zip`); PC 1's job removed from the jobs file so nothing double-places the exes | job `build-20261005-230745`, published 23:07:45Z (PC 2 woken), the first job on PC 2's re-set schedule; started 23:09Z, done 23:16:49Z (469 s): the Linux stage, the Windows stage 264 s, the test stage `RESULT test node [kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows] exit 0 54 s` and `RESULT test app/igneum-app [igneum-app] exit 0 6 s`; 9 outputs verified and placed: igneumd.exe be8e83c07aeae5eb6842768071735289f3ef149ed9a7088592d8c54a4c252c08 (51,321,856), igneum-miner.exe 1ba1a249e2a21d52087a81f61f37cd2e1e3273c7ec8809ef9cfaa98f015895ce (10,987,520), igneum-app.exe ab104cc0... (3,065,344, the PC's; the installer's engine is the runner's), Linux igneumd d7a2715e... (49,193,960, glibc 2.39, HiveOS), igneum-miner 09d05ff6..., igneum-app 222f30c0... |
| PC 2 suites | the combined job above (the PC 1 job of the row above never started: PC 1's app is down, section 3a) | six node suites exit 0 in 54 s, the app suite exit 0 in 6 s (23:16:49Z) |
| The Linux workers for HiveOS | `infra/cross/build-workers-linux.sh` (zig, glibc 2.36 target) under the lock, 23:10Z | igneum-worker-cuda 4aaff27fcb26bc5fe98d2b311b9f95f099c59414a556af32080b82b41e8360db (6,755,568), igneum-worker-opencl 82d90890be36f9b795b218794062e388bfd9c6f7524fe671074cdec84c906259 (306,000), ELF x86-64 dynamic; untested on a GPU host, as the script says |
| The HiveOS package | `NODE_OUT=<the zig node> WORKERS_OUT=<those workers> VERSION=0.3.11 packaging/hive/make-hive-package.sh`, 23:11:30Z, then `packaging/ota/publish-public.sh --hive` into `dl/public/` (the ship's deploy carries it) | `igneum-hive-0.3.11.tar.gz` c606a17043c013c411086b4abd0f38a227aa8a7db9d5934157760a3da5a5edc9 (24,496,653); no override inside (section 5) |
| The DMG | `NODE=<fork>/target-integration/release/igneumd MINER=... packaging/mac/build-dmg.sh` under the lock, 22:50:0x to 22:50:51Z | `Igneum-Miner-0.3.11.dmg` b7e81d4f6f3af9f9179e29faa72c844b795df757dfb2e4cd1d56cd454e78e1e7 (41,592,041): engine 0.3.11, node 89dfcb95 Mac arm64, the 0.3.9 prover host and export, `igneum-bench` (the Metal worker) rebuilt from this tree (667,145 to 555,808 bytes, signed ad hoc), fingerprints 477bb0ef and ed9c4d2e; read back from the mounted image: `Contents/Resources/igneum-app.json` carries the nine-field `node_override_params` with N4 = N5 = 154,800 (C34) |
### 3a. PC 1's app down, and the relay task that hit the wrong machine (22:31 to 23:05Z)
PC 1's installed 0.3.10 app quit at 22:31:06Z (its last upload 22:31:08Z, run `win-ae432dc7-20261005-214041`; Ember's tune job on it
reported "aborted (the app is quitting)"; the quit's sender is C35 for the consequences reviewer) and did not come back, so PC 1 mined
nothing, its build job `build-20261005-224654` could not start and no update-now could reach it. Three relay tasks (#241 at 22:52:31Z, #243 at
22:55:10Z, #245 at 23:03:18Z) went to the relay machine named "PC1" to relaunch the app, the second ending an engine that answered nothing on
`/api/state` and the third ending every igneum process before the launch, as the app's own updater does.
They hit the wrong machine. The relay's "PC1" is the 1ccfe586 box, the console's PC 2: both PCs carry the hostname DESKTOP-KMCV30N, the only
relay agent runs on the 1ccfe586 box and was named PC1 when the clients were set up, and the relay's PC2 entry reads "never seen". The
intake proves it: PC 2 got two new engine runs, `win-1ccfe586-20261005-225528` and `-230330`, at the exact times of #243 and #245, while PC 1
has no run after 21:40:41Z; #243's "hung" engine, pid 26696 from 21:49:40Z, was PC 2's healthy 0.3.10 engine (my probe's "no answer" on
`/api/state` was its own fault, no token), and the node, three miners and three workers it found were PC 2's own (a 5090 and the iGPU; PC 1
would have shown the 9070 XT too). So PC 2, the box the night's measurements run on, was force-restarted at 22:55:28Z and 23:03:30Z, which
killed the aggregation-cost agent's job 3 (re-run owed, 20 min) and ended whatever followed; its app came back each time (after #245: pid 30484,
responding, node, three miners and both workers up, the card climbing at 23:04Z). PC 1 is exactly as it was: engine down since 22:31:06Z,
no relay agent on that box, unreachable tonight; it waits for Josh in the morning and takes 0.3.11 through the manifest at its relaunch; the
fleet runs short its 141 MH/s until then. Told the Counter ASIC coordinator at 23:05Z; it re-set PC 2's schedule (this cut's combined build
and suite job first, then the prover-floor pair, the aggregation-cost re-run, "PC 2 clear", the M16 job) and ruled that no relay task goes to
"PC1" from anyone without its word. Every relay "PC1" reading tonight was PC 2 (the AMD agent's "9070 XT absent" probes read a box that has
no 9070 XT; the hardware events are being corrected). The relay machine should be renamed PC2 (`node tools/relay.mjs name <hostname> <name>`)
and the PC 1 box get its own agent (section 11). The Mac cross-build fallback for the Windows exes was announced and not started: the
combined PC 2 job replaced it within the minute.
## 4. The override object, N4 and N5, the digests
The object every node runs after the publish (the four live fields plus the five new ones):
```
{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000}
```
N4 and N5 are fixed BEFORE the DMG and the installer are built, because the packaged line (C34) must equal the manifest's object: DAA 136,967 at
22:45Z (PC 1's card) at 0.965 blocks/s puts the publish (about 23:40Z) near 140,200; tip + 14,400 is near 154,600; the first multiple of 3,600 at or
above it is 154,800 (N4's rule), which N5 takes too. At the publish the floor N - DAA >= 10,800 is checked; it holds until DAA 144,000 (about 00:45Z);
past that the line is re-pinned and the DMG and installer rebuilt.
The two digest readings on the 0.3.11 Mac node (bd7f043c..., ports 60975/60976, 22 s each, under `run`):
| Override file | Lines | Digest |
|---|---|---|
| none (the rolling-upgrade value: both new activations at never) | `igneumd/2.1.0-89dfcb95` | c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c, EQUAL to the node agent's pinned test on 79bd8e10 (22:49:59Z) |
| the nine-field object above | `Program class v3 from the override file: active from epoch 43 (DAA score 154800 rounded up to the epoch boundary at 154800, epochs of 3600 DAA)`, `Proving v1 from the override file: segment records paid from DAA score 154800, 8 blocks a segment, unproven after 600 DAA, aggregator share 1000 bps`, `Calibrated v1 fees ... 210000`, `Finality rule v3 ... 135200` | **0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888** (22:50:23Z): the value every node must print after the publish; the sweep in section 9 reads it on every node |
| the fleet's live four-field file (what every node runs today) | `Calibrated v1 fees ... 210000`, `Finality rule v3 ... 135200`, no class v3 or proving v1 line (both at never) | **4d8f8bb668828a3dcf7b783b995f3d3ebfde32a092dd1dbd5bf4373c5c65a62c** (23:13:54Z): the 0.3.11 BINARY alone flips the digest (the 0.3.10 node prints 1f4b4425... with the same file, `release-0.3.10.md` 9a): the new fields enter the digest even at never. This is the digest of step 1 (the reviewer's C40, the Counter ASIC rollout plan's section 4 step 1); both had assumed c562d70e..., which is the no-file case |
Three digests on the 0.3.11 binary, so two sweeps: step 1 moves every node to the binary (4d8f8bb6...) and step 2 moves every node to the
nine-field object (0139ab9d...). A node on either side of a sweep is refused by the other side (the handshake), so each sweep is one
window, as the fee switch's 8 min 47 s was.
## 5. The rollout order: two publishes, two sweeps (the reviewer's C39 and C40, the Counter ASIC coordinator's rule, the fee-switch shape)
Why two: the app writes the manifest's `consensus.override` at the manifest TAKE (`ota.rs` 661, `write_override`) and restarts its node with
it at the next safe window, whatever binary is installed; a 0.3.10 `igneumd` refuses a file with `program_class_v3_activation_daa` or the
proving v1 fields (`OverrideParams` is `deny_unknown_fields`) and dies at start, and the update then waits for a synced node (`engine.rs`
2211) until the slot minute or the 1,800-block rule forces it. One manifest with 0.3.11 AND the nine fields would take every 0.3.10 node
down at its next safe window: the 0.3.5 class the fee-switch plan named. And the 0.3.11 binary alone flips the digest (section 4), so the
binary move is itself a sweep.
| Step | What | Digest after |
|---|---|---|
| 1a | the observer, node 1 and the seed on the 0.3.11 binaries with the four-field file: `IGNEUMD=<fork>/target-integration/release/igneumd IGNEUMD_COMMIT=89dfcb95 infra/devnet/restart-hand-nodes.sh '<the four-field object>'`, then `IGNEUMD_LINUX=<the zig build> IGNEUMD_LINUX_SHA256=63cf490d... infra/devnet/restart-seed.sh '<the same>'`; the apps still on 0.3.10 are refused by them from this moment until each updates | 4d8f8bb6... on the three |
| 1b | the manifest: 0.3.11 with `consensus` carried over UNCHANGED (`--activation-height 135200 --deadline-note "finality v3"`, the four-field object, exactly as 0.3.10 shipped), `--public` (the HiveOS package rides along) | |
| 1c | update-now: the Mac (its card follows node 1) and the laptop first; PC 2 only on the Counter ASIC coordinator's "PC 2 clear"; PC 1 is down and unreachable (section 3a) and takes 0.3.11 through the manifest at its morning relaunch, refused until then. The watch: every app logs 0.3.11 and its node a DAA score at 4d8f8bb6...; every worker starts clean on the first try with the class-aware pair; the Mac's Metal worker takes the class v3 day from the prepared pack; C32: the agents whose kits sit on PC 2 republish their fetches after its update | 4d8f8bb6... on every reporting node |
| 2a | the floor: 154,800 minus the tip's DAA at least 10,800 (holds until DAA 144,000, about 00:45Z); past it N4 = N5 re-pinned to the first multiple of 3,600 at or above tip + 14,400, the packaged line, the DMG and the installer rebuilt; the DAA read sent to the Counter ASIC coordinator before 2b | |
| 2b | the hand nodes' and the seed's files switched to the nine-field object and restarted (the same two scripts), the manifest republished with `--override '<the nine-field object>' --activation-height 154800 --deadline-note "program class v3 + proving v1"`, update-now (the same order; PC 2 on "clear" again), the sweep | 0139ab9d... on every node |
| 3 | the plan's final sections, the merge to master (the live observer must not read stale: `public-api-check`'s other arm), the push, the report with per-machine times | |
The HiveOS package carries NO override: `packaging/hive/h-run.sh` line 31 starts the rig's node with `--devnet --appdir --rpclisten --listen` and
the peers, no `--override-params-file`, and no HiveOS package has ever carried one, so a rig's bundled node runs on genesis params and is refused
by every devnet peer (the HiveOS path is untested on a GPU host since 4 October). "Republish with the new override" therefore needs an
`h-run.sh` change (the file written from the Flight Sheet's extra config, as `PEERS=` is), which is the next cut's; tonight's package carries
the class-aware binaries only (section 11).
## 6. The push and CI
`git push -u origin release-0.3.11` at 3b0262f (23:18:29Z, the credential helper; the pre-push hook's site flip restored), `gh workflow run
windows.yml --ref release-0.3.11` -> run 37387737179, acquired at once (GitHub operational again), green 23:23:11Z (the parse job 23:18:39 to
23:19:25Z; engine, window host, payload, installer, smoke run 23:19:31 to 23:23:11Z, the G13 step against the 89dfcb95 inputs). The main
`ci.yml` run on 3b0262f (37387751432) failed in one step, the igneum-census release build (the class v3 fields missing from the census's own
initialisers, `fetch`'s new Layout argument, no Scratch/Hot match arms; pow tests, simulators and site jobs passed); the coordinator fixed it
in this worktree as 2a62735 (`igneum-census/src/main.rs` only; `git diff --stat 3b0262f 2a62735 -- app packaging igneum-pow proto-cuda
proto-opencl proto-metal` is empty, so the Windows artefacts of 37387737179 stand) and its run 37388453875 is green (pow tests and census
build, simulators, site). The 0.3.11 CI verdict is therefore run 37388453875 on 2a62735; the Windows build is run 37387737179 on 3b0262f,
the same app sources; the merge to master goes from 2a62735.
| File | sha256 | Size |
|---|---|---|
| Igneum-Miner-0.3.11.dmg | b7e81d4f6f3af9f9179e29faa72c844b795df757dfb2e4cd1d56cd454e78e1e7 | 41,592,041 |
| Igneum-Miner-Setup-0.3.11.exe | 84a21443f78598597f33cef307fa162e53c680dd90d29f02ceaf79ac5e231ee4 | 50,065,009 |
| igneum-windows-app.zip | ed0cf2a75a1de8897de33ddcdcdb4ea844eca6b38627d6b91c562700c9fbe5eb | 72,070,333 |
| igneum-hive-0.3.11.tar.gz (dl/public) | c606a17043c013c411086b4abd0f38a227aa8a7db9d5934157760a3da5a5edc9 | 24,496,653 |
## 7. The rollout, step 1 (the binary sweep to 4d8f8bb6...)
Baseline 23:19:40Z: tip DAA 139,642; the observer and node 1 on 21d4c73c at 1f4b4425...; the Mac app 0.3.10 (attached to node 1); PC 2
0.3.10 at 120.8 MH/s; PC 37ba0461 back on 0.3.10 at 2.3 MH/s; PC 1 down since 22:31:06Z (section 3a); Sam's Mac quit since 20:47Z.
| Step | Time | Result |
|---|---|---|
| 1a the observer | 23:23:54Z (pid 61754) | `igneumd/2.1.0-89dfcb95`, the four-field file, digest 4d8f8bb668828a3dcf7b783b995f3d3ebfde32a092dd1dbd5bf4373c5c65a62c |
| 1a node 1 | 23:24:06Z (pid 61864, caffeinate 61866) | the same |
| 1a the seed | 23:24:25Z (MainPID 125853) | `igneumd/2.1.0-89dfcb95`, the same lines; the a24ab01a... no: the 21d4c73c binary kept as `igneumd.prev-035` (the script's name) |
| 1b the ship (`--from ci`) | 23:24:44Z | preflight ok (tree 2a62735 clean, 0.3.11 in all 6, fork 89dfcb95, gh igneum-josh, live inputs 89dfcb95 built 23:17:51Z); ci, fetch, dmg, copy already; `consensus.override: carried over from the current manifest` (the four-field object), `activation_height` 135200, `deadline_note` "finality v3"; manifest 0.3.11 mac+windows signed (key 8f186e37...) and in `dl/public/`; one deploy; verify: the token folder's manifest 0.3.11, signature ok, mac b7e81d4f..., windows 84a21443...; the public folder's `igneum-downloads.json` not yet the local bytes at the edge (the 0.3.10 class), resumed from `verify` for the console item |
| 1c update-now, the Mac and the laptop | 23:28:06Z (`update-now-0311-d937c69d-37ba0461`, apps woken) | the Mac: the job ran 23:28:48Z, `Igneum Miner 0.3.11 is available: downloading (41 MB)` 23:28:49Z, engine restart 23:29:05Z (run `mac-d937c69d-20261005-232905`), 59 s after the job; `[ok] updated to Igneum Miner 0.3.11 from 0.3.10`; `a node already answers on 127.0.0.1:26610; using it`: the app attaches to node 1 (89dfcb95, 4d8f8bb6...), so its card follows node 1; `cards: Apple M5 Max [apple, off]`: its Metal miner was off before and after, so the prepared-pack check has no subject on the Mac tonight. The laptop (PC 37ba0461): (pending) |
| the console | 23:30Z | item #366 "Igneum Miner 0.3.11 shipped (mac+windows)" (`--from console` after the public index settled at the edge; the ship's own verify had refused it as 0.3.10's did) |
| the old side during the window | from 23:24Z | PC 2 and the laptop at 0 peers (refused by the new side) until each updates. The new side (node 1, the observer, the seed, the Mac app attached to node 1 with its miner off) has NO miner on it, so its chain STALLED at DAA 139,751 from about 23:29Z (the observer's `/api/stats` at 23:31:42Z: 139,751, age 1.8 s) until a miner joins it; the old side's fork grows only while a miner mines there, and PC 2's 0.0 MH/s from 23:29Z is the prover-floor sweep stopping its miners for its run (`floor-sweep-2`, started 23:21:17Z), not the refusal. Put to the Counter ASIC coordinator at 23:31Z with the numbers; its call: "PC 2 clear" the minute the sweep closes (about 23:36Z) and at 23:45Z at the latest, the restore job after the update, because an app restart under the sweep kills its job tree and its restore never runs; the laptop's 2 MH/s restarts the new side's chain when its install ends. The lesson for the next cut's plan: step 1a (the hand nodes and the seed first) moves the hub to a side with no hash until the first miner updates; the first update-now should go to a miner within the same minute |
| the Mac's miner (C42) | 23:39:10Z | the new side had NO miner: the Mac app's mining was a persisted `settings.paused` (cleared only by Resume; its STATUS lines read "0.00 MH/s, paused" since the update), node 1 runs no miner, the fleet's hash was on the refused old side. `POST <app.url>/api/resume` (the token path) answered ok at 23:39:10Z; the Metal worker compiled the program inline at 23:39:11Z ("not prepared": the prepared-pack path did not engage on this start, a G4b finding, section 11), raced 14 variants, 3.15 MH/s wall at 23:39:48Z; the new side's first blocks accepted 23:39:49, 23:39:51, 23:40:00Z. The stall: about 23:29Z to 23:39:49Z, 11 minutes of stopped DAA clock on the side every node ends up on. New check for step 1 of every digest-flipping cut: the new side has at least one miner before the first update-now |
| 1c update-now, PC 2 | 23:39:44Z (`update-now-0311-1ccfe586`, on the Counter ASIC coordinator's "PC 2 clear": `floor-sweep-2`'s first point had hung 18 min, nothing in flight to protect; its restore-and-diagnose job is the first PC 2 job after the update). PC 2 fetched the woken file at 23:40:13Z and logged `1 new for this machine, 1 queued`: update-now is itself a job and the app runs jobs one after another, so it waits behind the hung `floor-sweep-2` (started 23:21:17Z, cap 30 min, freed about 23:51:17Z); PC 2 reads `0.00 MH/s, waiting | node 139753 blocks, 0 peers, syncing` every 30 s meanwhile (refused by the new side, its miners stopped by the sweep). The class, for the next cut's plan: an update-now cannot pre-empt a running job; a hung job's cap sets the update's time | the queued job ran at 23:51:19Z, two seconds after the sweep's cap; installer verified 23:51:22Z; `quit: stopping the miners, then the node` 23:51:24Z; engine restart 23:51:30Z (run `win-1ccfe586-20261005-235130`), 11 min 46 s after the job and 3 min 35 s after PC 2 had received it; `igneumd started` 23:51:32Z; the prover's host and the pinned ids as before; `miner nvidia-1ccfe586-1 started` 23:51:43Z. Check 1 PASS at the first start: `epoch seed f4d9d3d8... (daa 140361): CPU program and cache ready in 166 ms`, `worker: ready cuda NVIDIA_GeForce_RTX_5090 ... first pack ... self-test PASS` at 23:52:21Z, no refusal. Then the hourly epoch boundary fell at DAA 140,401 (`SEED CHANGE ... f4d9d3d8... -> cb5b51cc...`, 23:52:21Z, 40 s after the first pack): the worker refused the new epoch's jobs (`epoch seed mismatch: this worker holds epoch f4d9d3d8..., the job is for epoch cb5b51cc...`, 80 lines over 80 s) while the app's attempt-aware path prepared the new epoch (`program pack checked for epoch seed cb5b51cc... day 20732: attempt 0`, the line the pack-loop fix added; `PREPARE sent ... class v2 (3559 DAA blocks before the boundary at 144000)`), the worker switched at 23:53:42Z, STATUS 113.4 MH/s wall at 23:54:14Z, the card 112.4 MH/s at 23:56:04Z. The Mac's worker crossed the same boundary in 241 ms (`prepared cb5b51cc... class v2`). PC 2's node joined the new side at once (its DAA tracks the observer's, 1 peer: node 1 at 192.168.68.64). The sweep's restore-and-diagnose job is the Counter ASIC coordinator's, after this |
| the laptop (PC 37ba0461) | silent since 23:28:28Z | its last upload, "2.25 MH/s, mining \| node 139757 blocks, 0 peers" at 23:28:28Z, is 22 s after `update-now-0311-d937c69d-37ba0461` was published; no job line reached the intake before the silence. Its 0.3.10 install kept it silent 55 minutes (21:41 to 22:36Z), so this is its install in progress until shown otherwise; publish 2 does not wait on it (a 2 MH/s machine whose 0.3.11 reads the nine fields when it returns; on 0.3.10 its node would die on the file until the forced apply, the C39 case for one machine) |
| PC 1 | unreachable tonight (section 3a); refused by every peer on 1f4b4425 until its morning relaunch takes 0.3.11 through the manifest | |
## 8. The rollout, step 2 (the object sweep to 0139ab9d...)
| Step | Time | Result |
|---|---|---|
| 2a the floor | 23:55:42Z | tip DAA 140,706; 154,800 - 140,706 = 14,094 >= 10,800, so N4 = N5 = 154,800 stand (the floor holds until DAA 144,000); no re-pin, no rebuild |
| 2b the observer | 23:56:59Z (pid 97249) | `/tmp/igneum-devnet/override-v3.json` switched to the nine-field object; `igneumd/2.1.0-89dfcb95`, `Program class v3 from the override file: active from epoch 43 (DAA score 154800 ...)`, `Proving v1 from the override file: ... 154800, 8 blocks a segment, unproven after 600 DAA, aggregator share 1000 bps`, digest 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 |
| 2b node 1 | 23:57:12Z (pid 97417) | the same lines and digest; it refused the seed (still 4d8f8bb6...) at 23:57:13Z and registered it at 23:57:42Z (protocol version 15), 29 s after the seed's own restart |
| 2b the seed | 23:57:30Z (MainPID 126124) | `/etc/igneum/override-v3.json` the nine-field object; the same binary 63cf490d..., the same lines, digest 0139ab9d... |
| 2b the manifest | 23:57:49Z | `publish-manifest.sh --version 0.3.11 --override '<the nine-field object>' --activation-height 154800 --deadline-note "program class v3 + proving v1" --notes '<section 1>' --public --deploy`; the live manifest's `consensus` read back byte-identical at the edge (the nine fields, `activation_height` 154800) |
| 2b update-now, the Mac and the laptop | 00:00:30Z (`update-now-0311-switch-d937c69d-37ba0461`, apps woken) | the Mac ran it 00:01:06Z: `0.3.11 is current`, `consensus parameters from the signed manifest: {... nine fields ...}`, `consensus override changed (.../Igneum/app/override.json); the node restarts with it at a safe moment`; the Mac's node card is node 1 (external: pid 0, starts 0, `consensus_digest` empty), already restarted by hand at 23:57:12Z, so there was nothing for the app to restart and its digest is node 1's. The laptop (PC 37ba0461) was not on the air (below) |
| 2b update-now, PC 2 | 00:01:10Z (`update-now-0311-switch-1ccfe586`, on the Counter ASIC coordinator's "PC 2 clear": `floor-restore-1` closed 23:57:29Z) | PC 2 fetched the woken file 00:01:41Z, ran the job the same second (`0.3.11 is current`, `consensus override changed (C:\Users\Admin\AppData\Local\igneum\app\override.json); the node restarts with it at a safe moment`), its node restarted at 00:01:42Z (`Consensus params digest: 0139ab9d...`, `igneumd/2.1.0`), the first STATUS 00:02:01Z "0.00 MH/s, waiting, node 141065 blocks, 1 peers, synced", mining at 00:02:31Z, 106.48 MH/s with 830 accepted this run and 0 faults at 00:06:31Z; the run id stays `win-1ccfe586-20261005-235130` (the node restarted, not the engine) |
| the window | 23:57:42 to 00:01:42Z | PC 2 (on 4d8f8bb6...) refused the seed (on 0139ab9d...) at 23:57:42Z and 23:58:12Z and mined on the old side alone; node 1 refused the seed once (23:57:13Z). The new side (the observer, node 1, the seed) had no miner on it either: the Mac's miner was off node 1 from 23:57:14Z (the next row), so the new side only relayed PC 2's old-side blocks it had already accepted (`PoW accepted ... daa 140757` at 23:57:32Z was the last) and waited; the two sides rejoined when PC 2's node restarted at 00:01:42Z and PC 2's hash carried the chain. The observer's tip read 140,757 at 00:02:22Z and 141,645 at 00:10:39Z (about 1.7 blocks/s, the catch-up after the rejoin), no stall as in step 1 |
| the Mac's miner (a miner finding) | 23:57:14Z to 00:08:34Z | the miner (`igneum-miner mine grpc://127.0.0.1:26610`) lost node 1 at node 1's 23:57:12Z restart and NEVER reconnected: 5,317 `submit error ... Not connected to server` and `template error: Not connected to server` lines, its TEMPLATES line frozen at `templates=2350` with `subscribed=true` while `fetch_errors` climbed (143 at 23:59:34Z, 595 at 00:05:07Z), the card's line "the node is not answering; the miner retries", the STATUS line still "26 MH/s, mining" with the accepted count frozen at 18,117 (773 this run). The retries are template and submit calls on a dead gRPC channel; nothing re-subscribes. Recovery through a job: `publish-jobs.sh add --kind restart --target d937c69d --what miners` (`restart-miners-0311-d937c69d`, 00:08:01Z); the Mac ran it 00:08:31Z (`miners restarted`), the new miner (pid 8567) started 00:08:33Z, its first block was accepted 00:08:34Z, the kernel race settled at 26.9 MH/s 00:09:09Z. Lost: 11 min 20 s of 26 MH/s. In step 1 this was masked: node 1's 23:24:06Z restart was followed by the Mac's own engine restart (the update) at 23:29:05Z, and the Mac was paused anyway |
| PC 2 at the epoch boundary | 23:52:22Z | 40 s after PC 2's first 0.3.11 pack the hour turned (`f4d9d3d8...` to `cb5b51cc...`): `hourly program changed: the pair was not prepared (unexpected seeds); the worker compiles inline`, 80 s of `worker fault: seed mismatch: the worker holds another program; preparing the current pair epoch cb5b51cc... day 2`, the worker on the new epoch at 23:53:42Z (113 MH/s), one `did not come back within 180 s` line for the old worker instance at 23:56:01Z. The restart landed inside the minute before the boundary, before the next epoch's prepare had run; the attempt-aware path (`program pack checked ... attempt 0`) recovered it without a job. Also at 23:52:14Z: `prover: block 74094 shard 0: ... Failed to create the CUDA prover impl: ... PermissionDenied (a GPU-server socket ...)`, the 0.3.10 socket class again on the first shard after the update; shards after it proved (`block 93665 shard 0 paid 0.9446373 IGN` at 00:06:09Z) |
| the laptop (PC 37ba0461) | absent | no upload since 23:28:28Z (section 7) and no card on the console at 00:08Z; both update-now jobs wait for it in the jobs file; its 0.3.10 app takes 0.3.11 and the nine-field object through the manifest when it is next on the air |
| PC 1 (ae432dc7), Sam's Mac (3a9bf309) | pending their relaunch | PC 1 down since 22:31:06Z (section 3a), Sam's Mac quit since 20:47Z on 0.3.9; each takes the manifest at its relaunch: the OLD app writes the nine-field object at the manifest take (`ota.rs` 661) and its old node, if restarted with that file before the 0.3.11 install lands, dies on the unknown fields (`deny_unknown_fields`, the reviewer's C39); the install then starts the 0.3.11 node on the file, so the worst case is one node death inside the update, to be read in the morning's intake |
## 9. The digest sweep
| Node | Binary | Digest | Since |
|---|---|---|---|
| the observer (`/tmp/igneum-devnet/observer-v4`, rpc 26640, json 28640) | `igneumd/2.1.0-89dfcb95` (bd7f043c...) | 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 | 23:56:59Z |
| node 1 (`/tmp/igneum-devnet/node1`, rpc 26610) | the same | 0139ab9d... | 23:57:12Z |
| the seed (188.245.5.161, `igneumd.service`) | `igneumd/2.1.0-89dfcb95` (63cf490d..., glibc 2.36 target) | 0139ab9d... | 23:57:30Z |
| PC 2 (1ccfe586, `devnet-v4`, rpc 26610) | the 0.3.11 installer's `igneumd.exe` be8e83c0... (PC 2's own build job) | 0139ab9d... | 00:01:42Z |
| the Mac (d937c69d) | attached to node 1 (no node of its own) | node 1's | 23:57:12Z |
| the laptop (37ba0461) | 0.3.10 node, 1f4b4425... at its last upload | pending | silent since 23:28:28Z |
| PC 1 (ae432dc7) | 0.3.10 node, 1f4b4425... | pending | down since 22:31:06Z |
| Sam's Mac (3a9bf309) | 0.3.9 node a24ab01a | pending | quit since 20:47Z |
The chain at 00:10:39Z: tip DAA 141,645 on the observer (age 1.4 s), node 1 with 2 peers (the seed and PC 2's side), PC 2 with 1 peer at
106 to 109 MH/s, the Mac at 26 MH/s, every live node on 0139ab9d... Epoch 43 (DAA 154,800) is about 3 h 45 min out at 0.965 blocks/s
(about 03:55Z); the class v3 and proving v1 lines of every node name it.
## 10. The next cut
| Branch | What | Why not 0.3.11 |
|---|---|---|
| `ember-tune` b671c8b (and the tune behind it) | the C35 fix: `Cmd::Quit(&'static str)` so every "quit:" line names its sender (the window host's stdin, the host gone, `POST /api/quit`, the sweep's end), `elevation_allowed()` = Power control alone (the unattended sweep on PC 1 raised one UAC prompt at 22:30Z under the old rule), no power cap at start under `--sweep`; the quit-source hunk is separable (main.rs 2 lines, server.rs 1 line, engine.rs the Quit arm, `elevation_allowed` and its test) | arrived after the tree closed at 23bc2b2 (the app, the DMG and PC 1's job carry it); not among the branches named for this cut |
| `fud-close` (the ledger closer's main branch, a22ba27a60a6f1c64; its ready tip was due about 23:05Z) | 45 public-text fixes on the site and litepaper, spec 8.3 and 8.8, two CI checks, relay fixes; touches `packfile.h` and `host.c`, so taking it means the two Windows workers, the Mac worker and the DMG rebuilt from the merged tip (G5) | offered by the Counter ASIC coordinator at 22:5xZ after the tree closed; not among the branches named for this cut; the fork-side `ledger-fixes` is not in 0.3.11 either |
| `explorer` d7e797c (and 3e01212) | `/api/stats` gains `proving`; `tools/ci/public-api-check.mjs` then FAILS when the live API lacks it, and ci.yml runs that check against the live site on every master push, which reads the OLD API until Vercel redeploys after the push (the reviewer's C36) | not in 23bc2b2 (only on the explorer branch); its merge needs the check to retry for a few minutes or to require `proving` only when `observer_updated_at` is newer than the commit |
| the app's job queue | an update-now job pre-empts a running job instead of queuing behind it, or the queue reports its wait in the STATUS line (tonight PC 2's update waited 11 minutes behind a hung sweep's cap, section 7; the Counter ASIC coordinator's ask) | app change |
| `proving-v1` c36dfea | docs only, after the code tip 22c2363 | its agent's choice: docs follow |
| the 0.3.10 list (`release-0.3.10.md` section 11): fork `pack-loop` 05ef0fa3, `job-console`, the rest of `opencl-rdna4`, `opencl-rdna4-telemetry` | | unchanged |
## 11. Open after the cut
| Item | What | Owner |
|---|---|---|
| the miner's dead channel (section 8) | `igneum-miner mine grpc://...` keeps `subscribed=true` after the node restarts and retries template and submit calls on the closed channel for ever (11 min 20 s tonight, 5,317 errors, hash burned at 26 MH/s with 0 accepted); it must re-subscribe on the first `Not connected to server` and the app's card must read "the miner lost the node" instead of "mining, 26 MH/s". Until the fix: every hand restart of node 1 is followed by `publish-jobs.sh add --kind restart --target d937c69d --what miners` (the runbook's step), and the Mac's card is read by its accepted count, not its MH/s | miner (next cut) |
| the HiveOS package carries no override | `packaging/hive/h-run.sh` starts the rig's node without `--override-params-file`; a rig on `igneum-hive-0.3.11.tar.gz` runs on genesis params and is refused by every peer; the file must come from the Flight Sheet's extra config as `PEERS=` does (section 5) | packaging (next cut) |
| the prepared pack and a restart near the hour (section 8) | an engine restarted in the last minute before an epoch boundary has no prepared pair for the next hour (80 s of seed-mismatch faults on PC 2 at 23:52:22Z, recovered by the attempt-aware path); the prepare should run at engine start for the next epoch too, or the update-now path should wait out the boundary when it is under 2 min away | app |
| PC 2's first shard after an update | `Failed to create the CUDA prover impl: ... PermissionDenied (a GPU-server socket ...)` on block 74094 at 23:52:14Z, the 0.3.10 socket class (fixed 22:01Z for the running engine; the root socket outlives the engine restart); the later shards proved. The socket unlink belongs in the engine's prover start, not only in the playbooks (`prover-socket-check.sh` covers the playbooks) | app / proving |
| the relay machine named PC1 (section 3a) | it is the 1ccfe586 box: rename it (`node tools/relay.mjs name DESKTOP-KMCV30N PC2`), give the ae432dc7 box its own agent, and no relay task to "PC1" without the Counter ASIC coordinator's word until then | relay (Josh in the morning for the PC 1 box) |
| PC 1 down since 22:31:06Z | relaunch in the morning (the installed app, double-clicked or through the Start menu, never elevated); it takes 0.3.11 and the nine-field object through the manifest, one node death inside the update possible (section 8); the intake's first lines of `win-ae432dc7-...` are the check; `ember-tune` b671c8b carries the C35 fix for the quit's sender | Josh, then the release engineer |
| the laptop (37ba0461) | silent since 23:28:28Z, off the console; its 0.3.10 install kept it silent 55 min once already; when it uploads again its first STATUS line must show 0.3.11, the nine-field object in the manifest log line and 0139ab9d... | watch |
| Sam's Mac (3a9bf309) | 0.3.9, quit since 20:47Z; `min_supported` is 0.3.0 so it updates straight to 0.3.11 at its relaunch | watch |
| the app's job queue | an update-now queues behind a running job (PC 2's step-1 update waited 11 min behind a hung sweep's 30 min cap, section 7); pre-empt, or report the wait in the STATUS line | app (section 10) |
| the public index at the edge | `dl/public/igneum-downloads.json` alternates between the unsigned and the signed copy at the edge for minutes after a deploy, so the ship's verify refuses it once per cut (0.3.10 and 0.3.11 both); `--from console` after it settles is the workaround; the verify should accept either copy while both carry the cut's hashes, or the deploy should purge the edge | ship-app |
| the pre-push site flip | the pre-push hook rebuilds `site/` and leaves the tree modified (`git checkout -- site/` after every push); the site build is not idempotent | site |
| the explorer branch's strict check (C36) | `public-api-check.mjs` on `explorer` d7e797c fails against the live API until Vercel redeploys; merge it right after a master push, not before | explorer (section 10) |
| C1 (the reviewer) | the 0.3.10 `exportSegments` lacks `daaScore` and `feesV1ActivationDaa`; proving-v1's `rpc.rs:982` (eb32c645) carries them; the decision on which the aggregator reads is due 16:00Z 6 October | the Counter ASIC coordinator |
| the Metal worker at resume (G4b) | "the pair was not prepared" at the Mac's 23:39:10Z resume (section 7), compiled inline; same class as the PC 2 boundary row above | app |
| the 0.3.10 list | `release-0.3.10.md` section 11, unchanged: fork `pack-loop` 05ef0fa3, `job-console`, the rest of `opencl-rdna4`, `opencl-rdna4-telemetry`, the per-job zip pin check | |

View file

@ -164,7 +164,7 @@ Every emitted instruction satisfies: `rot` in 1..31, `mask` in {1, 2, 4, 8, 16},
### 1.4.5 Encoding
A program is transmitted as the seed bytes, never as instructions. A node hands a miner the pack it emits itself (`igneum-pow/src/emit.rs`: `kernel.cu`, `kernel.cl`, `program.metal`, `program.h`, `program.json`, `memhard.h`, `vectors.*`, the three `*_bound` kernels), and a miner MAY regenerate everything from the seed bytes by the procedure of 1.4.6. `program.json` (format `igneum-program-pack-3`) is the interchange form; its field names are those of `Instr` in `generator.rs`, and it carries `generator` (2), `attempt`, `program_id` and `seed_bytes`. `program.h` carries the same as `IGNEUM_GENERATOR`, `IGNEUM_PROGRAM_ATTEMPT`, `IGNEUM_PROGRAM_ID` and `IGNEUM_SEED_BYTES_HEX`. An implementation MUST refuse a pack whose generator version is not its own.
A program is transmitted as the seed bytes, never as instructions. A node hands a miner the pack it emits itself (`igneum-pow/src/emit.rs`: `kernel.cu`, `kernel.cl`, `program.metal`, `program.h`, `program.json`, `memhard.h`, `vectors.*`, the three `*_bound` kernels), and a miner MAY regenerate everything from the seed bytes by the procedure of 1.4.6. `program.json` (format `igneum-program-pack-3`) is the interchange form; its field names are those of `Instr` in `generator.rs`, and it carries `generator` (2), `attempt`, `program_id` and `seed_bytes`. `program.h` carries the same as `IGNEUM_GENERATOR`, `IGNEUM_PROGRAM_ATTEMPT`, `IGNEUM_PROGRAM_ID` and `IGNEUM_SEED_BYTES_HEX`. An implementation MUST refuse a pack whose generator version is not its own. Program class v3 (Counter ASIC 2.0, 5 October 2026, activated by the height switch `program_class_v3_activation_daa` from the first epoch whose start score is at or above it, `docs/plans/counter-asic-2-rollout.md`) writes `generator` 3, and every pack of it carries `IGNEUM_PROGRAM_CLASS` (`v3`) and `IGNEUM_ERA_SEED_HEX` (the 32-byte era seed of section 1.13.1, or its devnet stand-in) beside `IGNEUM_GENERATOR`; the serve protocol's `prepare` and `job` lines carry `class=v3 era=<hex>` for v3 epochs and nothing for v2 ones. A worker MUST refuse a pack whose class or era seed does not match the line it was prepared for (`igneum-pow/src/packcheck.rs`, `verify_pack_dir_chain`; `proto-cuda/nvrtc/packfile.h`), and a pack of a generator other than 2 or 3.
### 1.4.6 Program acceptance
@ -178,7 +178,7 @@ Implemented (`igneum-pow/src/accept.rs`, `proto-metal/main.swift`; ledger M6 Fix
Attempts. Attempt 0 of a program seed `b` (the 32-byte epoch seed, or the UTF-8 of a seed string) is the candidate drawn from `seed_words_from_bytes(b)`. If it fails, attempt `k = 1, 2, ...` is drawn from `seed_words_from_bytes(b || k_le32)`; the first accepted candidate is the program of the epoch. Measured rejection rate under this generator: 5.14 percent over 100,000 seeds (census section 7) and the 20,000-seed confirmation of `docs/bench-log.md` (4 October 2026), so the probability that 32 consecutive candidates fail is below 2^-136, and an implementation MAY treat 32 consecutive failures as a consensus fault (`MAX_ATTEMPTS`).
Program id. `FNV-1a-64("igneum-program/" || generator_le32 || seed words as little-endian bytes || attempt_le32)` with `generator = 2`, written into every pack. Two implementations that agree on the id agree on the generator version, the seed words and the attempt.
Program id. `FNV-1a-64("igneum-program/" || generator_le32 || seed words as little-endian bytes || attempt_le32)` with `generator = 2` under class v2 and `generator = 3` under class v3, written into every pack. Two implementations that agree on the id agree on the generator version, the seed words and the attempt.
Why the closed form: the test is then a pure function of the program (no cache, no day), costs 1.3 to 3.4 ms on one core, and the census checked on 100,000 programs that its verdict agrees with the memory-hard dataset's on all but 39 threshold-edge cases (section 7.3). What the three parts catch: (a) the empty-list fallback of 1.4.3; (b) registers that saturate to all ones (2.4 percent of candidates); (c) zero-absorbing register sets, lane-constant load sites, output bias and value-level address repeats (2.1 percent). Not in the rule, and why: a contraction as the last write (80 percent of programs) and the `or` count are too common and (c) already catches the cases that matter; the load critical path is a hash-rate question, not a weakness.
@ -207,7 +207,7 @@ Implemented for the prototype size; Designed for genesis.
A load reads one 4-byte word at `src AND MASK`. Every load in every emitted kernel has exactly this form; the static check in `TESTS.md` section 5 is part of conformance (section 1.15). Because item values do not depend on the dataset size (section 1.8.5), the 1 GiB vectors remain valid for words below 2^28 at any larger size.
Growth beyond genesis is in section 1.13.
Growth beyond genesis is in section 1.13; under program class v3 the cache follows the dataset's doublings (1.13.3) and the verifier holds 256 MiB, then 512 MiB from year 4 and 1 GiB from year 12.
## 1.6 Register initialisation
@ -319,7 +319,21 @@ s = M_8(s)
item(t) = s
```
Eight dependent cache reads (`ITEM_ROUNDS = 8`, prototype value): the address of read `r` depends on every earlier read. Nine mixer applications. `dataset[w] = item(w >> 4)[w AND 15]`. A dataset of 2^D words is the prefix of items `0 .. 2^(D-4) - 1`, so an item has the same value at every dataset size.
Eight dependent cache reads (`ITEM_ROUNDS = 8`, prototype value): the address of read `r` depends on every earlier read. Nine mixer applications under program class v2.
Program class v3 (Counter ASIC 2.0, decided 5 October 2026, delegated; Josh confirms for the public testnet genesis) applies the mixer `m = 8` times per round with distinct round keys (`LoadClass::mixer_mult`; `docs/plans/mixer-x4.md` section 2), the eight dependent reads unchanged:
```
for r in 0..7:
for j in 0..m-1:
s = M(s, rk = (r * m + j + 1) * 0x9E3779B9)
a = s[0] AND (2^(C - 4) - 1) cache line index, 2^(C - 4) lines of a 2^C-word cache (C from 1.13.3)
s[i] = s[i] XOR cache[line a][i] for i in 0..15
for j in 0..m-1:
s = M(s, rk = (8 * m + j + 1) * 0x9E3779B9)
```
Under `m = 1` the keys are `(r + 1) * 0x9E3779B9` and `9 * 0x9E3779B9`, the class v2 text exactly; the `9 m` keys are the first `9 m` values of the sequence `k * 0x9E3779B9`, all distinct. Why `m = 8`: the recompute attacker's cost is operations per item (`docs/analysis/m16-recompute-attacker-2026-10-05.md`); the honest miner pays the mixer once a day in the dataset build, which stays latency-bound (RTX 5090 23 to 25 ms, RX 9070 XT 72 to 77 ms, M5 Max 21 ms at `m` = 1, 4 and 8, measured 5 October 2026); the verifier pays `m` per item it derives: 0.61 ms per warp at `m = 1`, 1.24 at 4, 2.08 at 8 on one loaded M5 Max core (measured 5 October 2026, `docs/plans/mixer-x4.md` 6.4a), inside the 10 ms gate. The on-die-cache recompute chip's gain against the RTX 5090 falls from 2.4x (`m = 1`) to 0.92x with a 3x fixed-function factor at `m = 8` (`docs/analysis/chip-model-v3.md`, approximate factor). Under class v3 an item's value also depends on `C` through the line mask, so the items change on the day the cache doubles (1.13.3); the emitted `mh_item` carries the `m` loop only for `m > 1`, so every class v2 pack keeps its text. Class v3 also draws the dataset layout and the load windows per era (`docs/plans/era-layout.md`; the strided windowed load address and the interleaved mapping `mh_addr`, one text form in the three dialects). `dataset[w] = item(w >> 4)[w AND 15]`. A dataset of 2^D words is the prefix of items `0 .. 2^(D-4) - 1`, so an item has the same value at every dataset size.
What the construction buys (Measured, `MEMHARD.md` section 2.2, M5 Max, seed igneum-genesis, 1 GiB):
@ -396,11 +410,11 @@ All times are DAA seconds since genesis (section 0.6). At 1 block per second one
| Clock | Length | What changes | Label |
|---|---|---|---|
| Epoch | 3,600 DAA s | The program: new seed words from the VDF of section 4, new kernel | Designed (design document, "Always evolving, on three clocks"); the epoch length is a prototype value, to be fixed at gate 2 by the difficulty-tracking measurement (fork map c1: the hash rate steps by program, 35 to 48 Mhash/s across seeds on the M5 Max, so the DAA window must track within an epoch) |
| Epoch | `epoch_len(d)` DAA s, base 3,600; the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200; set by 90% miner signal at a day boundary (sections 1.13.1 and 5.7) | The program: new seed words from the VDF of section 4, new kernel | Designed (Counter ASIC 2.0 layer 9, 5 October 2026, `docs/plans/epoch-length.md`); 3,600 stays the value on every network until a signal moves it, and stays the prototype value to be fixed at gate 2 by the difficulty-tracking measurement (fork map c1: the hash rate steps by program, 35 to 48 Mhash/s across seeds on the M5 Max, so the DAA window must track within an epoch) |
| Day | 86,400 DAA s | The day key, hence the cache and the dataset | Designed |
| Era | 15,552,000 DAA s (180 days) | Era parameters and one instruction-family unlock, section 1.13 | Designed; the length is a prototype value (the design says "every 6 months") |
Epoch `e` covers DAA scores `[3,600 e, 3,600 (e + 1))`. The epoch of a block is the epoch of its own DAA score, so "which program was this block mined under" is a function of the header alone once the seed is known. The program for epoch `e` is `generate_from_seed_bytes(program_seed_e)`: attempt 0 is drawn from `S_e = seed_words_from_bytes(program_seed_e)`, and a rejected attempt is replaced as 1.4.6 says; `program_seed_e` is the 32-byte VDF output of section 4.3.
Epoch `(d, e)` covers DAA scores `[86,400 d + L e, 86,400 d + L (e + 1))` with `L = epoch_len(d)` and `e` in `0 .. 86,400 / L`; every ladder step divides 86,400, so day boundaries are epoch boundaries, and the epoch is identified by its start score `s = 86,400 d + L e`. At the base `L = 3,600` this is `[3,600 e, 3,600 (e + 1))` and nothing below differs from the earlier text. The epoch of a block is the epoch of its own DAA score, and `epoch_len(d)` is a function of the blue blocks of the signalling window that closed at least 2 days before day `d` (section 1.13.1), which are in the header's past, so "which program was this block mined under" is a function of the header alone once the seed is known. `T_epoch` and the 1,200-s lead of section 4.3 are genesis constants and do not follow `epoch_len`: the program of every epoch is known 600 s before it starts on the reference core at every length. The program for epoch `e` is `generate_from_seed_bytes(program_seed_e)`: attempt 0 is drawn from `S_e = seed_words_from_bytes(program_seed_e)`, and a rejected attempt is replaced as 1.4.6 says; `program_seed_e` is the 32-byte VDF output of section 4.3.
Implementation note (devnet, 3 October 2026, `docs/fork-divergence.md` "Epoch seed"): until the VDF of section 4 is in the node, `program_seed_e` is the hash of the last selected-chain block whose DAA score is below `3,600 e - 600`. The 600-DAA-score lead stands in for section 4.3's 20-minute lead: the program of epoch `e` is knowable about 10 minutes before it starts, every block template reports it (`pow_epoch.next_epoch_seed`), and a GPU worker compiles it in the background and swaps at the boundary with no pause (serve protocol `prepare`, `proto-metal/main.swift`, `proto-cuda/host.cu`, `proto-opencl/host.c`). Measured across boundaries on a short-epoch test network in `docs/bench-log.md` (hot-swap entry). The program schedule is a protocol constant; a miner that cannot compile ahead sees the same seed at the same time as everyone else, only later.
@ -422,12 +436,31 @@ The era seed `E_n` is the 32-byte output of the 1-hour VDF of section 4.4. One S
| Load count | 16 of 64 | not drawn | fixed, so every era is equally memory-bound |
| Output fold rotations | (7, 14, 21), (9, 18, 27) | each `1 + below(31)` | 1..31 |
| Mixer round count | 8 | not drawn | fixed, so the verify budget holds |
| Mixer applications per round `mixer_mult` | 4 (class v3; 1 under class v2) | not drawn | fixed at genesis (Counter ASIC 2.0, 5 October 2026: the M16 recompute chip's only measured lever; section 1.8.5 carries the form; `docs/plans/mixer-x4.md`) |
| Epoch length `epoch_len` | 3,600 DAA s | one draw of the era stream consumed and not used (the value is set by signal) | the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 (layer 9, `docs/plans/epoch-length.md`) |
`epoch_len` is the one era-table parameter set by miners rather than by the draw: 90% of blue blocks over a 7-day window carrying the same ladder index (3 bits of the header version, encoding Open in section 5.8) sets that length from the first day boundary at least 2 days after the window closes (section 5.7). It is not a code upgrade: the rule, the ladder and the window are genesis constants, and the chain carries no release. The era stream consumes its draw so that a future draw of this parameter changes no other parameter's value. The threat it answers is a per-program hard datapath (an FPGA fleet: 42 to 160 minutes per compile on a mid-size part, PRflow, FPT 2019, hours on large parts; at 600 s nothing it compiles ever runs); it does not answer a programmable chip, which the other layers answer. The floor 600 is set by the slowest compile-ahead measured (the Metal variant race, 38 s on the M5 Max, 6.3% of a 600-s epoch and inside the 600-s seed window; `docs/plans/epoch-length.md` section 6).
The table layout and the working-set window (Counter ASIC 2.0 layers 4 and 8, decided IN on 5 October 2026, delegated: the six-era hash-rate spread is 1.3% on the RTX 5090, 3.2% on the RX 9070 XT and 0.8% on the M5 Max, under the 5% rule; `docs/plans/era-layout.md`) are drawn under program class v3 by a second stream `S` seeded with words 0 and 1 of `seed_words_from_bytes("igneum-era/" || E_n)` (the index is not in the preimage: `E_n` commits to `n` through the VDF input), seven draws in this order whether or not a value is used:
1. `W = allowed[below(|allowed|)]`: the width in words of every dataset load of the era, from the genesis-fixed set `allowed`; the set is `{1}` (4 bytes, the read-width decision of 5 October 2026), so the draw is consumed and the width pinned.
2. `M = low32(next()) OR 1`: the stride multiplier, odd, so `x -> x * M` is a bijection.
3. `R = 1 + below(31)`: the stride rotation.
4. to 7. `r_i = next()` for `i` in 0..3: the interleave draws. With `b = log2(W)` and `free = 4 - b`, `c = [b, ..., 15]`; for `i` in `0..free`: `j = i + (r_i mod (16 - b - i))`, swap `c[i]` and `c[j]`; the interleave is `pos = [0, ..., b - 1] ++ sort(c[0..free])`, four ascending bit positions below 16.
The era parameters are `(W, M, R, pos)`. Dataset mapping under class v3: word `w` holds word `j(w)` of item `t(w)`, where bit `i` of `j(w)` is bit `pos[i]` of `w` and `t(w)` is `w` with bits `pos[0..3]` removed; with `pos = [0, 1, 2, 3]` this is `dataset[w] = item(w >> 4)[w AND 15]` byte for byte; an item keeps its value at every dataset size of at least 2^16 words, and the `W` words of one aligned load lie in one item, so the 4,096-item verifier bound of 1.11 holds. Load address under class v3, for a load site with window draws `(k_off, o)` and a dataset of `2^D` words: `k = min(k_off, D - 26)`, `y = rotl(x * M, R)`, `idx = ((y AND (MASK >> k)) OR ((o AND (2^k - 1)) << (D - k))) AND MASK` (uniform on the window, branch-free, three operations before the mask), one text form in Metal, CUDA and OpenCL. The window draws per instruction (layer 8), after the nine draws of 1.4.3: `k_off = below(3)` (the dataset, a half or a quarter) and `o = low32(next()) AND (2^k_off - 1)`, used only on a load slot, so a class v3 program takes 720 draws; the window never goes below 2^26 words (256 MiB, above the largest on-chip cache in the benchmark) nor above the dataset, and sixteen sites with drawn offsets cover the dataset with high probability (a windows-union census over 300 programs: the SRAM mirror a chip would need is the whole dataset in every hour). The acceptance rule of 1.4.6 is unchanged in its tests and mirrors this address at its constant `D = 28`. Devnet stand-in for `E_n` until the VDF of 4.4 is in the node: era 0 the genesis block hash; era `n >= 1` the hash of the last selected-chain block whose DAA score is below `15,552,000 n - 7,200`. What the interleave buys and does not: a chip that hard-wires one layout reads the wrong 15 words with every word once the era draws another; a chip whose address decoder can permute its address lines pays nothing (stated in the plan). The stride is a bijection with no cryptanalysis yet (Open).
"Memory pattern" in the design document is read here as the item-address pattern (the cache line index word, `s[0]` in 1.8.5, and the XOR-all-sixteen rule); the proposal is to leave it fixed at era 0 and let the unlocked families change the kernel instead, because every change to the item derivation changes the verify time and must be re-measured.
### 1.13.2 Instruction-family reserve
At genesis the generator carries the eleven families of 1.4.1 live and a reserve list of further families in a fixed order. At the start of era `n >= 1`, reserve family `n` becomes live with weight `W_new` taken proportionally from the live non-load families. A family may enter the reserve only if it is integer-exact and has passed the cross-vendor conformance of section 1.15 on every vendor in the benchmark (Metal, CUDA, OpenCL on NVIDIA and AMD), with its own edge-case vectors, before genesis. Candidate families, all integer ALU operations present on Apple, NVIDIA and AMD: variable left shift and logical right shift by `src AND 31`; bit-field extract with an immediate offset and width; `andn` (`dst = dst AND NOT src`); byte permute of `dst` by an immediate selector; population count and count-leading-zeros folded into `dst` by add; a three-register select (`dst = bit of src2 ? src : dst`); a second shuffle form (`lane + delta mod 32`). The order and `W_new` are Open. A family that is not in the genesis reserve can only be added by the upgrade path of section 5.7.
At genesis the generator carries the eleven families of 1.4.1 live and a reserve list of further families in a fixed order. At the start of era `n >= 1`, reserve family `n` becomes live with weight `W_new` taken proportionally from the live non-load families. A family may enter the reserve only if it is integer-exact and has passed the cross-vendor conformance of section 1.15 on every vendor in the benchmark (Metal, CUDA, OpenCL on NVIDIA and AMD), with its own edge-case vectors, before genesis. Candidate families, all integer ALU operations present on Apple, NVIDIA and AMD: variable left shift and logical right shift by `src AND 31`; bit-field extract with an immediate offset and width; `andn` (`dst = dst AND NOT src`); byte permute of `dst` by an immediate selector; population count and count-leading-zeros folded into `dst` by add; a three-register select (`dst = bit of src2 ? src : dst`); a second shuffle form (`lane + delta mod 32`). The order and `W_new` are Open, except the first entry, decided 5 October 2026 (Counter ASIC 2.0 layer 7, delegated; Josh confirms for the public testnet genesis; `docs/analysis/int8-matrix-family.md`):
> Reserve family R1, `mm8` (integer matrix). Semantics: section 2.2 of `docs/analysis/int8-matrix-family.md`, uint8 operands from `src` and `src2` in the m8n8k16 fragment layout, one int32 element of C per lane selected by the immediate `bit`, added into `dst` modulo 2^32. Weight at unlock `W_new = 4` points, taken proportionally from the ten live non-load families (the load weight and count are untouched). Edge vectors, each a hand-built unit run on every vendor: all bytes 0xFF in A and B (C = 1,040,400 everywhere); all bytes 0x80 (C = 262,144); A all zero (C = 0); `dst` = 0xFFFFFFFF with a nonzero C (the wrap); alternating 0x00 and 0xFF by lane; `bit` = 0 and 1 on the same fragments. Unlock: at the start of era n = 4 (DAA 62,208,000), or earlier by the 90% signalling path of section 5.7; never by a release. Native paths: PTX `mma.sync` `.u8` (sm_75+), AMD WMMA `i32_16x16x16_iu8` (RDNA 3 and 4), Metal 4 `mpp::tensor_ops::matmul2d` (`uchar x uchar -> int`); the per-lane `dot4` form is emulation on Apple (1.6x per op unsigned, measured 5 October 2026) and is not the reserved form.
A vendor that can only emulate. A family enters the reserve when it is bit-exact on every vendor of 1.15. A vendor that reaches the result only by emulation (no instruction or library path) does not block entry if the measured penalty of the emulation on that vendor, on the family's own probe (a dependent chain of the op against the same vendor's integer ALU chain), is at most 8x per op, AND the family's weight at unlock keeps the emulating vendor's hash-rate loss under 5% on the memory-hard hash, checked on the vendor's card with the family live. A family whose emulation exceeds either bound stays out of the reserve until the vendor ships a path.
A family that is not in the genesis reserve can only be added by the upgrade path of section 5.7.
### 1.13.3 Dataset growth
@ -437,7 +470,7 @@ Designed: 2 GiB at genesis plus 0.5 GiB per year (design document, "Which cards
N_d = floor((2 GiB + 0.5 GiB * (86,400 d / 31,536,000)) / 64 bytes)
```
evaluated in integers (bytes), with one year = 31,536,000 DAA seconds. The dataset grows by about 23 KiB per day and is recomputed with the day key. Two consequences are Open:
evaluated in integers (bytes), with one year = 31,536,000 DAA seconds. The dataset grows by about 23 KiB per day on this average and is recomputed with the day key. Decided 5 October 2026 (Counter ASIC 2.0 layer 6, delegated; Josh confirms for the public testnet genesis): the cache grows with the dataset, doubling when the dataset doubles: `cache_log2_words(d) = 26 + growth_doublings(d)`, `growth_doublings(d) = floor(log2(1 + d / 1,460))` for day `d` since genesis (doublings at years 4, 12, 28 and 60), so the cache is 256 MiB at genesis, 512 MiB from year 4, 1 GiB from year 12; the verifier's one-core fill is 0.2, 0.4 and 0.8 s at those steps (0.2 s per 256 MiB, section 1.12), under 1 s at every step of the schedule. The dataset steps to the next power of two on the same doublings (option (b) below, recommended to Josh with the card-lifetime consequences in `docs/analysis/card-lifetime-2026-10-05.md`: a 4 GB card mines to year 4, an 8 GB card to year 12, a 12 GB card to year 28 with the cache freed after the daily build). Why the cache grows at all: an SRAM mirror of a flat 256 MiB cache is about 128 mm^2 and $46 of silicon at N5 by shipped cache-die density (AMD V-Cache, 64 MB on 41 mm^2 at 7 nm; `docs/analysis/sram-mirror.md`), so the cache size never prices a chip out; its job is to stay above any GPU's on-die cache (96 MB on the RTX 5090, 128 MB on GB202), which a flat 256 MiB loses within the decade. The GPU frees the cache after the daily dataset build; the hash never reads it. Two consequences remain Open:
- Index mapping. `src AND MASK` requires a power-of-two size. For a non-power-of-two `N_d` the proposed mapping is `idx = (src * N_words) >> 32` computed in 64 bits (a multiply-shift range reduction; uniform to within 2^-32, branch-free, integer only). At `N_words = 2^28` this gives `src >> 4`, not `src AND MASK`, so adopting it changes the 1 GiB vectors; gate 1 chooses between (a) the multiply-shift mapping with new vectors, or (b) power-of-two sizes only, growing in steps (2 GiB, 4 GiB) on the same schedule's average, which keeps `AND MASK` and means a 4 GiB card lasts until the 4 GiB step instead of fading.
- The item index `t` is 32 bits, so the construction as written tops out at 2^32 items = 256 GiB, which the schedule reaches after 508 years. No action needed.
@ -509,4 +542,6 @@ Pack `proto-cuda/packs/igneum-devnet-v4-epoch0/`: epoch seed bytes `edc4fa844da9
Header-bound vectors (section 1.6 rule, seed `igneum-genesis`, day `2026-10-03`): `igneum-pow/README.md`, eight values, for example H = 32 zero bytes and nonce 0 give `746c567b090acf6a`.
Program class v3 vectors (5 October 2026, `proto-cuda/packs-ca2-mixer/` and `proto-cuda/packs-ca2-era/`; generator 3, mixer x8, the cache growth rule, the era draw inside the class): pack `mx8-genesis` (seed `igneum-genesis`, no era, program id `e323b9dcaf283a6f`, batch fingerprint `7c28cfb06c5c65a9`) and pack `mx8-devnet-epoch0` (the devnet epoch seed and day of 1.17 with the era stand-in E_0 = the devnet genesis hash inside the class, program id `73bcbfe8ccf988f1`, unit 0 lane 0 `d424577fce4a7a60`, batch fingerprint `90f794dd556f7a3b` over 2^24 outputs at base nonce 0), reproduced by the Rust interpreter, Metal and Apple OpenCL on 5 October 2026 (3/3 standalone, 3/3 in batch, 96 of 96 lanes each); the six era packs `era-0` to `era-5` (the same epoch seed and day, era test seeds 0 to 5, program id `73bcbfe8ccf988f1`; 2^24 fingerprints `8e8e070db4eea52d`, `891c01b8563bb47e`, `e54279fed2831b5d`, `77e0ba8abbd0ae62`, `d898d8f4f2e7684b`, `a6927db380f7efb2`). All seven fingerprints are equal on the RTX 5090 (CUDA/NVRTC), the RX 9070 XT (AMD OpenCL) and the M5 Max (Metal), self-test PASS on every pack, and 1,024 random nonces per card on `era-0` and `mx8-devnet-epoch0` re-hash to the same value on the Rust verifier (1,024 of 1,024 each): job `run-ca2-era-pc1b-20261005`, 5 October 2026, `docs/bench-log.md`. The class v2 vectors above stand unchanged (the v2 exports are byte-identical on the class v3 crate, `igneum-pow/tests/packs.rs`).
Cache, dataset and mixer vectors: section 1.8.4 and 1.8.5. Seed words: section 1.3.1. Generator: section 1.4.3. Acceptance: section 1.4.6. Batch fingerprints: section 1.15 item 5.

View file

@ -49,7 +49,7 @@ Designed. Epoch e is the DAA-score interval `[3,600 e, 3,600 (e + 1))` (section
1. **Seed checkpoint.** `C(e)` is the highest-index checkpoint (section 3, C1) whose checkpoint block has DAA score at most `3,600 e - 1,200`: the latest checkpoint at least 20 minutes of DAA time before the epoch starts. The checkpoint block hash is Kaspa's full header hash, which covers the nonce, as the grinding defence requires (`proto-vdf/README.md`: a hash that covers only the body would let a miner start the VDF while still searching nonces).
2. **Evaluation.** `input = hash(C(e))`, `T = T_epoch`, run 4.2. `program_seed_e` is the 32-byte output; `proof_e` is (T, y, pi).
3. **Program.** `S_e = seed_words_from_bytes(program_seed_e)` (section 1.3.1), program = `generate_from_words(S_e)` (section 1.4). The 1,200-s lead is 2x the reference evaluation time, so a core half as fast as the reference still finishes before the epoch (4.6).
3. **Program.** `S_e = seed_words_from_bytes(program_seed_e)` (section 1.3.1), program = `generate_from_words(S_e)` (section 1.4). The 1,200-s lead is 2x the reference evaluation time, so a core half as fast as the reference still finishes before the epoch (4.6). The lead and `T_epoch` are fixed whatever the epoch length of section 1.12: at the floor of 600 DAA s the checkpoint is two epochs back and the program is known one full epoch ahead; at the base it is known for the last sixth of the previous epoch (`docs/plans/epoch-length.md`, section 3).
4. **Header.** Every header carries `seed_source = hash(C(e))` for its own epoch (section 2.4). A header is valid under the lottery only if its `seed_source` is a block on its own selected chain at the blue score of checkpoint index `i(C(e))`, and its PoW verifies under the program derived from that block. The proof is not in the header: a node verifies `proof_e` once per epoch (4.5 ms) and caches `S_e`.
Determinism of step 1 is the point of the rule: which block is "the checkpoint at blue score 30 i" is a function of the header's own past, so two nodes validating the same header derive the same program, and a header mined under a reorged-away checkpoint names a block that is not on its chain and is invalid. Whether `C(e)` must be certified (section 3) or merely be the selected-chain block at that blue score in the header's past is Open (O-4.3): requiring certification couples mining to finality liveness (a stall longer than the lead would stop the program from being derivable), which the design document accepts ("an epoch cannot start without a valid proof") and this specification argues against, because the chain is meant to keep running on plain GHOSTDAG through a finality pause (section 3.7 item 2). The proposal: the selected-chain block at that blue score, certified or not, deep enough (1,200 DAA s plus d) that a reorg across it is a merge-depth-scale event.

View file

@ -36,6 +36,8 @@ Self-dealing: a developer who also mines the including block collects 80% plus 2
Designed. The 20% emission share (section 2.5) and the provers' part of the 80% tip share are paid per block as a fixed amount for that block, divided among the block's shards by consensus proving cost, so a stuffed block earns no more than an honest one. Shards are not claimed first-come and carry no bond: each shard is assigned by sortition to 8 eligible provers for a 10-s exclusive window, then open to anyone, and the first valid proof included in a block is paid (section 7.2, decided 3 October 2026, ledger P8, C9). The parameters 8 and 10 s are set on the phase 4 devnet (O-5.1). A withheld shard costs nothing to bond against because nothing waits on an assigned prover: an unproven block delays only its proof; execution and the 30-s lock do not wait for it (ledger P9). The bond, slashed on a bad or late proof, remains in the external job market (5.4), where a customer does wait; its size and timeout are Open (O-5.6).
Proving v1 (section 7.8, 5 October 2026, Implemented behind `proving_v1_activation_daa`): from the switch, `proving_v1_aggregator_share_bps` of a block's fixed amount (a tenth, Josh's decision at 0.3.11) goes to the aggregator whose segment record attests the block, the rest to the shards as before; a block in a segment that stays unproven past `proving_v1_unproven_daa` pays no aggregator share.
## 5.4 External job market
Designed. At launch external proving jobs are paid on the customer's chain, in the customer's currency, to a payout contract keyed by miner address, because Igneum cannot yet see Ethereum; the customer chain's own bond and slashing apply (design document, "The first six months"). When the job market settles on Igneum, which needs the proof bridge (phase 2 consensus proof, ledger P4, E7), every job fee paid in IGN splits:

View file

@ -78,6 +78,13 @@ Nothing in consensus changes for any of this: the segment claim already commits
| Records per block | 8 | Implemented, 7.7 (Designed value) |
| Payout per shard | the segment's pool credit in equal parts, remainder to shard 0, paid by the carrying segment | Implemented, 7.7 |
| Proving v0 activation | `proving_v0_activation_daa`, default never | Implemented, 7.7 |
| Segment record (v1) | per segment of `proving_v1_segment_blocks` chain blocks, 586 bytes (the aggregator guest's 340-byte statement inline), BLS-signed by the aggregator's vote key, in the coinbase extra data before the shard record section (`IGNS`); proof bytes on p2p message 75 (protocol 15) | Implemented, 7.8 (branch `proving-v1`, 5 October 2026) |
| Segment records per block | 2 | Implemented, 7.8 (Designed value) |
| Segment length `N` | `proving_v1_segment_blocks`, 8 | Implemented, value Decided (5 October 2026, delegated; `docs/plans/proving-v1.md`) |
| Unproven deadline `T` | `proving_v1_unproven_daa`, 600 DAA s after the segment's last chain block | Implemented, value Decided (5 October 2026, delegated) |
| Aggregator share | `proving_v1_aggregator_share_bps`, 1,000 (a tenth of every attested block's pool credit; the shards share the rest) | Implemented, value Decided (5 October 2026, delegated) |
| Proving v1 activation | `proving_v1_activation_daa`, default never; the segment grid starts at the first chain block at or above it | Implemented, 7.8 |
| Mandatory proofs | the rule of 7.8 item 9, no switch yet, off | Designed |
## 7.5 Proving gas per transaction: the cap and the abort
@ -112,3 +119,20 @@ Added 4 October 2026 because the implementation (`vendor/igneum-node-proving`, b
8. **The plan.** Every segment has at least one shard; an empty segment is one shard whose statement applies the rewards and payouts only. The node cuts from its own per-transaction boundaries (`TxBoundary`: the carry of 7.6, the state root after every transaction) with the cut of `igneum_prove_core::plan`; `igneum-prove-export` must reproduce the node's plan on the same export, and the test network checks that it does.
RPCs (the execution layer's JSON-RPC): `igneum_getShardPlan(block)`, `igneum_getProofRecords(block)`, `igneum_submitProofRecord({record, proof})`, `igneum_getAssignedShards([keyHash...], lookback)` (the prover's work list), `igneum_getProvingStatus()`. Signing without the BLS key material in the prover process: `igneum-miner sign-record` and `key-hash`.
## 7.8 Segment records, the chain rule and the unproven rule, as implemented (proving v1)
Added 5 October 2026 (Josh: "open the proving round asap"; branch `proving-v1` of the fork and of the main repository, `docs/plans/proving-v1.md`). Every item is Implemented on the branch and behind `proving_v1_activation_daa` (default never); nothing here changes the devnet until the 0.3.11 rollout sets the switch. Item 9 is Designed and off. Proving v0 (7.7) keeps running underneath: per-shard records stay valid and paid.
1. **The aggregated proof.** The aggregator guest of 7.6 (pinned, `elf/igneum-prove-aggregator`) verifies every shard proof of one chain block and, by recursion, the previous chain block's aggregated proof (`AggInput.prev`): its public values (`BlockOutput`, 340 bytes, mirrored in consensus as `BlockStatement`) carry `chain_len`, the number of consecutive chain blocks the proof attests, and `agg_vk`, the aggregator's own id whenever `chain_len > 1`. One compressed proof of constant size therefore attests any run of consecutive chain blocks (measured: `docs/bench-log.md`, "proving v1, aggregated chains on the RTX 5090"). This is design 5.3's "segment N verifies N-1" realised inside the proof rather than beside it.
2. **Segments.** From the first chain block `A` whose DAA score reaches `proving_v1_activation_daa`, chain blocks are grouped in fixed segments of `N = proving_v1_segment_blocks`: segment `k` is `A + kN ..= A + kN + N - 1`. The grid is a pure function of the chain, so every node names the same segments.
3. **The record.** One `SegmentRecord` (`kaspa_consensus_core::proving`): version 2, `first`, `last`, the hash of chain block `last`, the aggregator's BLS vote key, the payout address, the aggregated proof's 340 public values inline, the SHA-256 of the proof bytes, and a BLS signature over all of it under `IGNEUM_SEGMENT_RECORD_V1`, domain-separated with the network name. 586 bytes. The statement the verifier checks is keccak256 of the public values.
4. **Carriage.** Records ride in the coinbase extra data as a section `records || len_le32 || "IGNS"` placed before the shard record section (`IGNP`), which sits before the finality section; at most 2 per block. A node that does not read the section sees miner bytes. The proof bytes travel beside the record on p2p message `IgneumSegmentRecordMessage` (type 75, protocol version 15; peers at 14 and below never receive it; 14 is the EVM transaction relay of 0.3.10) and through `igneum_submitSegmentRecord`.
5. **What every node checks on a carried record (consensus).** The segment is aligned on the grid and its last block is on the executor's own chain, at most 600 chain blocks behind the carrier; the signature verifies; the public values equal the node's native block statement for chain block `last` in every field but `provers` and `chain_len` (`chain_id`, `number`, `block_hash`, `parent_hash`, `shard_count`, `tx_commitment`, `pre_root`, `post_root`, the keccak of the shard receipts roots, gas, pgas, executed, skipped, `shard_vk` = the pinned shard program id, `agg_vk` = the pinned aggregator id when `chain_len > 1` and zero otherwise); `chain_len` is at least `N` and at most the chain height; the chain rule of item 6; and the deadline of item 7. A record that fails is ignored, not a fault of the block: it pays nothing. The SP1 proof is not verified on this path (item 8).
6. **The chain rule.** A record whose `chain_len > N` chains to the previous segment's proof by recursion (the proof itself verified it), and is valid whatever the chain says about that segment. A record whose `chain_len = N` starts a fresh chain and is valid only for the first segment of the grid, or when the previous segment is unproven (item 7). So a proven segment is followed only by records that verify it; an unproven one may be skipped.
7. **The unproven rule.** A segment whose last chain block is more than `T = proving_v1_unproven_daa` DAA seconds old at the carrier, with no paid record, is unproven: the pool pays nothing for it (its aggregator share stays in the escrow), a record for it carried after the deadline is invalid, and the next segment may start a fresh chain. A segment with a paid record is proven; before its deadline it is pending. `igneum_getProvingStatus` reports the three counts over the record window.
8. **Verification is off the consensus path**, as 7.7 item 4: the proof pool verifies segment proofs through `igneum-prove-host --mode verify-segment` (SP1's light verifier against the pinned aggregator key; the shard id and the aggregator id inside the statement checked against the pinned ids; keccak of the public values against the statement) and a producer offers only verified records to its templates. The native statement bounds what an unverified record can do to the aggregator's payout, never to state.
9. **Payout.** From the activation, a chain block's pool credit (5.3) splits: `proving_v1_aggregator_share_bps` of it to the aggregator of the segment record that attests the block, the rest to its shards in equal parts as 7.7 item 6 (`split_pool_credit`). At the carrying chain block, for every segment record its blocks carry in sequence order, the first valid record per segment pays the sum of the aggregator shares of the segment's blocks to the record's payout address, from the escrow, after the shard payouts and before the transactions; a reorg unwinds it with the carrier. There is no aggregator sortition yet: the first valid record carried wins (Open, O-7.3: the VRF draw of design 5.3).
10. **Mandatory proofs (Designed, off).** The rule that makes a proof a condition of validity: a chain block is invalid if the segment ending `T` DAA seconds before it is unproven. It needs an activation height of its own (not yet a parameter) and a consensus-level check in the block validator; it is written here so the devnet measures the coverage the fleet can meet first (`docs/plans/proving-v1.md`, step 3) and Josh sets the height when the share is one.
RPCs: `igneum_submitSegmentRecord({record, proof})`, `igneum_getSegmentStatement(block)` (the segment, its status, the native public values for a fresh and a continuing chain, the shard proofs the pool holds per block, the previous paid record), `igneum_getSegmentRecords(block)`, `igneum_getProofBytes(block, shard, keyHash)` and `igneum_getSegmentProofBytes(first, keyHash)` (the aggregator's inputs), the `v1` object of `igneum_getProvingStatus`. Signing: `igneum-miner sign-segment-record`. Host modes: `chain`, `aggregate`, `verify-segment`.

View file

@ -20,7 +20,8 @@ use std::sync::Mutex;
use std::time::Instant;
use igneum_pow::accept::{check as accept_check, Reject};
use igneum_pow::generator::{candidate_from_words, generate_v1_from_words, GeneratorConfig, Instr, Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LANES, OP_WEIGHTS};
use igneum_pow::generator::{candidate_from_words, generate_v1_from_words, GeneratorConfig, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LANES, OP_WEIGHTS};
use igneum_pow::memhard::Layout;
use igneum_pow::seed::{fnv1a64, program_rng, seed_words, SplitMix64};
use igneum_pow::verify::{dataset_elem, hash_warp, splitmix32, DatasetMode, DatasetSource};
@ -180,9 +181,9 @@ fn generate_fixed16(seed_string: &str, seed: [u32; 8], fresh: Fresh) -> Program
fresh_reg[src as usize] = false;
}
fresh_reg[dst as usize] = true;
instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask });
instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width: 1, win: 0, off: 0 });
}
Program { seed_string: seed_string.to_string(), seed_bytes: seed_string.as_bytes().to_vec(), seed, generator: GENERATOR_VERSION, attempt: 0, instrs }
Program { seed_string: seed_string.to_string(), seed_bytes: seed_string.as_bytes().to_vec(), seed, generator: GENERATOR_VERSION, attempt: 0, class: LoadClass::V2, era_bytes: None, instrs }
}
// ---------------------------------------------------------------------------------------------------------
@ -478,6 +479,9 @@ fn run_warp(p: &Program, base: u32, ds: &DatasetSource, acc: &mut Acc, lane_addr
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
// Experiment-only ops of the read-width and scratch variants; the census generates the lottery
// hash's class only, so neither op ever appears here.
Op::Scratch | Op::Hot => unreachable!("the census never generates a scratch or hot op"),
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
@ -505,7 +509,7 @@ fn run_warp(p: &Program, base: u32, ds: &DatasetSource, acc: &mut Acc, lane_addr
}
match memhard {
Some(m) => {
m.fetch(&idx, &mut val);
m.fetch(&idx, &mut val, Layout::LINEAR);
}
None => {
for lane in 0..LANES {

View file

@ -11,12 +11,14 @@
//! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) |
//!
//! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and
//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the
//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by
//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the
//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES};
use crate::seed::{fnv1a64, SplitMix64};
use crate::verify::{dataset_elem, splitmix32};
use crate::memhard::hot_index;
use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel};
/// Units (32-lane warps) the dynamic test interprets.
pub const ACCEPT_UNITS: usize = 64;
@ -33,6 +35,14 @@ pub const BIAS_TOLERANCE: u32 = 136;
/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
pub const MIN_DISTINCT_SUM: u64 = 245_760;
/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
pub fn min_distinct_sum(loads: usize) -> u64 {
loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
}
/// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the
/// first failing test in the order of the module table.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
@ -67,7 +77,7 @@ impl std::fmt::Display for Reject {
Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
Reject::DistinctAddresses { sum } => {
write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120)", *sum as f64 / 2048.0)
write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
}
}
}
@ -170,6 +180,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
let seed = &p.seed;
let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1;
let (d0, d1) = (seed[0], seed[1]);
let (h0, h1) = (seed[2], seed[3]);
let hot_words = p.hot_words();
let loads = p.loads_per_hash();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
@ -183,12 +195,30 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
let mut idx = [0u32; LANES];
let mut nload = 0usize;
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
let slot_mask = p.class.scratch_slot_mask();
let era = p.class.era;
for it in 0..ITERATIONS {
let sel = r[0];
for (k, ins) in p.instrs.iter().enumerate() {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Scratch => {
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
for lane in 0..LANES {
idx[lane] = r[a][lane] & slot_mask;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]);
lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane];
}
nload += 1;
}
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
@ -255,18 +285,45 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
}
Op::Load => {
// Read-width experiment: a load of `width` words reads from the aligned address and folds every
// word (verify::fold_words); width 1 is the lottery hash's xor of one word.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = r[a][lane] & mask;
idx[lane] = load_index(era.as_ref(), ins, r[a][lane], mask, ACCEPT_DATASET_LOG2) & align;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
if width == 1 {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
} else {
let mut w = [0u32; 16];
for j in 0..width {
w[j] = dataset_elem(idx[lane] + j as u32, d0, d1);
}
r[d][lane] = fold_words(r[d][lane], &w[..width]);
}
lane_addrs[lane * loads + nload] = idx[lane];
}
nload += 1;
}
Op::Hot => {
// Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with
// bit 30 so a hot word and a dataset word at one index count as two addresses.
for lane in 0..LANES {
idx[lane] = hot_index(r[a][lane], hot_words);
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], h0, h1);
lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane];
}
nload += 1;
}
Op::WLoad => {
let b = (r[a][0] & mask) & !31;
for lane in 0..LANES {
@ -298,7 +355,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
sl.sort_unstable();
let mut distinct = 0u64;
for k in 0..loads {
if k == 0 || sl[k] != sl[k - 1] {
// scratch slots carry bit 31 (variant 5) and are not dataset addresses
if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
distinct += 1;
}
}
@ -333,7 +391,7 @@ pub fn check_dynamic(p: &Program) -> Result<AcceptReport, Reject> {
}
bias_max = bias_max.max(d);
}
if acc.distinct_sum <= MIN_DISTINCT_SUM {
if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
}
Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
@ -348,9 +406,50 @@ pub fn check(p: &Program) -> Result<AcceptReport, Reject> {
#[cfg(test)]
mod tests {
use super::*;
use crate::generator::{candidate, generate, GeneratorConfig, generate_v1};
use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass};
use crate::verify::{DatasetMode, DatasetSource};
#[test]
fn distinct_bound_scales_with_the_load_count() {
assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM);
assert_eq!(min_distinct_sum(32), 61_440);
}
/// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees
/// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
#[test]
fn classes_pass_and_match_verify() {
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-rw-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
let bases = accept_base_nonces(&p.seed);
let loads = p.loads_per_hash();
let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut ones = [0u32; 64];
for (u, &b) in bases.iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for h in crate::verify::hash_warp(&p, b, &ds) {
for j in 0..64 {
ones[j] += ((h >> j) & 1) as u32;
}
}
}
assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter");
}
}
/// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words.
#[test]
fn instrumented_interpreter_matches_verify() {
@ -381,6 +480,52 @@ mod tests {
}
}
/// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are
/// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads).
#[test]
fn hot_classes_pass_and_hot_loads_are_uniform() {
for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-hot-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
}
let p = generate_class("igneum-genesis", LoadClass::hot(96, 4));
let words = p.hot_words();
let loads = p.loads_per_hash();
let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut buckets = [0u64; 16];
let mut hot_count = 0u64;
for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for &a in &la {
if a & 0xC000_0000 == 0x4000_0000 {
let idx = a & 0x3FFF_FFFF;
assert!(idx < words);
buckets[(idx as u64 * 16 / words as u64) as usize] += 1;
hot_count += 1;
}
}
}
assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes");
let mean = hot_count as f64 / 16.0;
for (i, &b) in buckets.iter().enumerate() {
assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}");
}
// the dataset distinct count still holds for the dataset loads alone
let r = check(&p).unwrap();
assert!(r.distinct_mean() > 120.0);
}
#[test]
fn base_nonces_are_aligned_and_seed_dependent() {
let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]);

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -32,7 +32,7 @@ pub mod verify;
pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256};
pub use accept::{check as accept_program, AcceptReport, Reject};
pub use generator::{generate, generate_from_seed_bytes, Instr, Op, Program, GENERATOR_VERSION};
pub use memhard::{Cache, MemhardCpu, MixParams};
pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, V3_CLASS};
pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape};
pub use seed::{fnv1a64, seed_words, SplitMix64};
pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch};

View file

@ -7,11 +7,22 @@
//! [--epoch-hex <64 hex> --day-hex <hex>] byte seeds instead of strings (Epoch::from_seed_bytes)
//! igneum-pow accept --seed <s> [--epoch-hex <64 hex>] every candidate of the seed with its verdict (spec 01 section 1.4.6)
//! igneum-pow show --seed <s> [--epoch-hex <64 hex>] the accepted program, one instruction per line
//!
//! Read-width experiment (5 October 2026, docs/plans/read-width.md): `--class v2|w4|w16|w64|w64x4|p4,p16,p64[xN]`
//! on every command selects the load class (default v2, the lottery hash). Nothing in a v2 run changes.
//!
//! Era layout (5 October 2026, docs/plans/era-layout.md): `--era igneum-era-test/<n>` (a test era seed: the 32 bytes
//! of seed_words_from_bytes of the string, era index n) or `--era <n>:<64 hex>` (the chain's 32-byte era seed E_n)
//! turns the chosen class into its era class; `--era-widths 4` (default: the read-width decision of 5 October 2026
//! keeps v2's 4-byte load) is the allowed width set the era draws from; `4,16,64` lets the era draw the width.
use igneum_pow::generator::{EraParams, GENERATOR_VERSION_V3};
use igneum_pow::emit::export_pack;
use igneum_pow::memhard::Cache;
use igneum_pow::generator::{LoadClass, ProgramClass};
use igneum_pow::memhard::{Cache, Shape};
use igneum_pow::seed::day_key;
use igneum_pow::verify::{DatasetMode, Epoch, DEFAULT_DATASET_LOG2};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2};
use std::time::Instant;
struct Args {
@ -26,6 +37,51 @@ struct Args {
prehash: String,
epoch_hex: Option<String>,
day_hex: Option<String>,
class: LoadClass,
/// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache).
days: u64,
/// The program class (Counter ASIC 2.0 seam): v2 (default) or v3, which draws from V3_CLASS with generator 3.
program_class: Option<ProgramClass>,
/// The era seed bytes a class v3 chain program records (`--era-hex`).
era_hex: Option<String>,
/// Era layout: `--era igneum-era-test/<n>` or `--era <n>:<64 hex>` composes the era class over `--class` with
/// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout).
era: Option<(u64, Vec<u8>, String)>,
era_widths: Vec<u8>,
}
/// `igneum-era-test/<n>` or `<n>:<64 hex>` -> (index, 32 era bytes, label).
fn parse_era(s: &str) -> Option<(u64, Vec<u8>, String)> {
if let Some(n) = s.strip_prefix("igneum-era-test/") {
let index: u64 = n.parse().ok()?;
return Some((index, EraParams::test_era_bytes(s).to_vec(), s.to_string()));
}
let (n, hex) = s.split_once(':')?;
let index: u64 = n.parse().ok()?;
let bytes = igneum_pow::bind::unhex(hex)?;
if bytes.len() != 32 {
return None;
}
Some((index, bytes, format!("igneum-era/{index}/{hex}")))
}
/// "4,16,64" (bytes) -> ascending words.
fn parse_widths(s: &str) -> Option<Vec<u8>> {
let mut v: Vec<u8> = s
.split(',')
.map(|x| match x.trim() {
"4" => Some(1u8),
"16" => Some(4),
"64" => Some(16),
_ => None,
})
.collect::<Option<Vec<_>>>()?;
v.sort_unstable();
v.dedup();
if v.is_empty() {
return None;
}
Some(v)
}
fn usage() -> ! {
@ -36,7 +92,12 @@ fn usage() -> ! {
\x20 hash --nonce <n> print the 64-bit hash of one nonce (pack form, init words = seed words)\n\
\x20 hash-bound --prehash <64 hex> --nonce <u64> print the header-bound hash (bind.rs) of one 64-bit nonce\n\
\x20 accept every candidate of the seed (or --epoch-hex) with its acceptance verdict\n\
\x20 show the accepted program, one instruction per line"
\x20 show the accepted program, one instruction per line\n\
\x20 --class C load class: v2 (default), mx4 (class v3: mixer x4, cache growth), w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g]\n\
\x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\
\x20 --program-class v2|v3 the program class of the seam (v3 = generator 3 on V3_CLASS, the chain's own derivation; --era-hex records the era seed)\n\
\x20 --era E era layout over --class: igneum-era-test/<n> or <n>:<64 hex> (the 32-byte era seed E_n)\n\
\x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)"
);
std::process::exit(2)
}
@ -54,6 +115,12 @@ fn parse() -> Args {
prehash: "00".repeat(32),
epoch_hex: None,
day_hex: None,
class: LoadClass::V2,
days: 0,
program_class: None,
era_hex: None,
era: None,
era_widths: vec![1],
};
let mut it = std::env::args().skip(1);
a.cmd = it.next().unwrap_or_else(|| usage());
@ -70,12 +137,30 @@ fn parse() -> Args {
"--prehash" => a.prehash = val(),
"--epoch-hex" => a.epoch_hex = Some(val()),
"--day-hex" => a.day_hex = Some(val()),
"--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()),
"--days" => a.days = val().parse().unwrap_or_else(|_| usage()),
"--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())),
"--era-hex" => a.era_hex = Some(val()),
"--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())),
"--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()),
_ => usage(),
}
}
if let Some((_, bytes, _)) = &a.era {
a.class = LoadClass::era(a.class, bytes, &a.era_widths);
}
a
}
/// An era program is a class v3 program: generator 3 and the era bytes recorded (what the chain's
/// `Epoch::from_chain_seeds` does); the pack then carries IGNEUM_PROGRAM_CLASS "v3" and IGNEUM_ERA_SEED_HEX.
fn stamp_era(e: &mut Epoch, a: &Args) {
if let Some((_, bytes, _)) = &a.era {
e.program.generator = GENERATOR_VERSION_V3;
e.program.era_bytes = Some(bytes.clone());
}
}
fn main() {
let a = parse();
let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard };
@ -85,21 +170,14 @@ fn main() {
"accept" => accept(&a),
"show" => show(&a),
"hash" => {
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
let (e, _) = epoch_of(&a, mode);
println!("{:016x}", e.hash(a.nonce as u32));
}
"hash-bound" => {
let bytes = igneum_pow::bind::unhex(&a.prehash).unwrap_or_else(|| usage());
let prehash: [u8; 32] = bytes.as_slice().try_into().unwrap_or_else(|_| usage());
// --epoch-hex / --day-hex: the chain's byte seeds (Epoch::from_seed_bytes), as the worker protocol carries them
let e = match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
Epoch::from_seed_bytes(&eb, &db, "cli")
}
_ => Epoch::new(&a.seed, &a.day, mode, a.dataset_log2),
};
let (e, _) = epoch_of(&a, mode);
let init = igneum_pow::bind::block_init_words(&prehash, a.nonce);
println!("init words {}", init.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" "));
println!("{:016x}", e.hash_bound(&prehash, a.nonce));
@ -108,6 +186,49 @@ fn main() {
}
}
/// The epoch every command works on, and the day label for packs. `--epoch-hex`/`--day-hex` give the chain's byte
/// seeds (the day label then names the day bytes); else the string seed and day. `--program-class v3` draws the
/// program through the seam (generator 3 on `V3_CLASS`, the era bytes of `--era-hex` recorded) and sizes the
/// dataset for `--days` through `Epoch::chain_dataset_day`; `--class` is ignored under a program class (the class
/// names the load class). Closed-form mode is only for string seeds under the default class.
fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let (mut e, label) = epoch_of_class(a, mode);
stamp_era(&mut e, a);
(e, label)
}
fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let era = a.era_hex.as_ref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage()));
match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
let label = format!("igneum-epoch/{eh}/day/{dh}");
let e = match a.program_class {
Some(pc) => Epoch {
program: Epoch::chain_program(&eb, era.as_deref(), pc, &label),
dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2),
},
None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2),
};
(e, format!("bytes:{dh}"))
}
_ => {
let e = match a.program_class {
Some(pc) => {
let program = igneum_pow::generator::generate_from_seed_bytes_program_class(&a.seed, a.seed.as_bytes(), pc, era.as_deref());
let lc = pc.load_class();
let shape = Shape::for_class_day(&lc, a.days);
let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 };
Epoch { program, dataset: DatasetSource::new_shape(&a.day, mode, log2, shape) }
}
None => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days),
};
(e, a.day.clone())
}
}
}
fn bench(a: &Args, mode: DatasetMode) {
println!(
"igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})",
@ -116,21 +237,41 @@ fn bench(a: &Args, mode: DatasetMode) {
a.dataset_log2,
mode.name()
);
let shape = Shape::for_class_day(&a.program_class.map(|pc| pc.load_class()).unwrap_or(a.class), a.days);
if mode == DatasetMode::MemoryHard {
// Time the cache fill on its own first (one core), then build the epoch (which fills it again).
let t0 = Instant::now();
let c = Cache::fill(day_key(&a.day));
let c = Cache::fill_log2(day_key(&a.day), shape.cache_log2_words);
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("cache: fill {fill_ms:.1} ms on one core (2^26 words, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", c.fnv1a64());
println!(
"cache: fill {fill_ms:.1} ms on one core (2^{} words, {} MiB, {} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}",
shape.cache_log2_words,
shape.cache_words() * 4 / (1 << 20),
c.segments(),
c.fnv1a64()
);
drop(c);
}
if let Some(h) = a.class.hot {
// the hot table of the epoch on its own first (one core), then the epoch (which fills it again)
let t0 = Instant::now();
let t = igneum_pow::memhard::HotTable::for_seed_bytes(a.seed.as_bytes(), h.mb as u32);
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("hot table: {} MiB filled in {fill_ms:.1} ms on one core ({} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", h.mb, igneum_pow::memhard::hot_segments(h.mb as u32), t.fnv1a64());
}
let t0 = Instant::now();
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
let (e, _) = epoch_of(a, mode);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!(
"program: {} loads/hash, {} items/warp, op mix {}; epoch built in {build_ms:.1} ms",
"program: class {}, {} loads/hash, {} bytes/hash, widths (1,4,16 words) {:?}, {} items/warp, mixer x{} ({} mixers/item), cache 2^{} words, op mix {}; epoch built in {build_ms:.1} ms",
e.program.class.name(),
e.program.loads_per_hash(),
e.program.bytes_per_hash(),
e.program.width_counts(),
e.program.items_per_warp(),
shape.mixer_mult,
shape.mixers_per_item(),
shape.cache_log2_words,
e.program.op_mix()
);
let bases = [0u32, 4096, 1_000_000];
@ -158,14 +299,7 @@ fn export(a: &Args, mode: DatasetMode) {
let out = a.out.clone().unwrap_or_else(|| usage());
let t0 = Instant::now();
// --epoch-hex / --day-hex: the chain's byte seeds; the day label then names the day bytes
let (e, day_label) = match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
(Epoch::from_seed_bytes(&eb, &db, &format!("igneum-epoch/{eh}/day/{dh}")), format!("bytes:{dh}"))
}
_ => (Epoch::new(&a.seed, &a.day, mode, a.dataset_log2), a.day.clone()),
};
let (e, day_label) = epoch_of(a, mode);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("igneum-pow export {out}");
println!(
@ -179,7 +313,7 @@ fn export(a: &Args, mode: DatasetMode) {
e.program.program_id(),
e.program.loads_per_hash()
);
println!("op mix: {}", e.program.op_mix());
println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts());
let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name());
let pack = export_pack(&e, &day_label, &source);
let dir = std::path::Path::new(&out);
@ -209,7 +343,7 @@ fn seed_bytes_of(a: &Args) -> (String, Vec<u8>) {
fn accept(a: &Args) {
let (label, bytes) = seed_bytes_of(a);
let t0 = Instant::now();
let tries = igneum_pow::generator::attempts(&label, &bytes);
let tries = igneum_pow::generator::attempts_class(&label, &bytes, a.class);
let ms = t0.elapsed().as_secs_f64() * 1e3;
for (p, verdict) in &tries {
match verdict {
@ -233,19 +367,42 @@ fn accept(a: &Args) {
fn show(a: &Args) {
let (label, bytes) = seed_bytes_of(a);
let p = igneum_pow::generator::generate_from_seed_bytes(&label, &bytes);
let mut p = igneum_pow::generator::generate_from_seed_bytes_class(&label, &bytes, a.class);
if let Some((_, eb, _)) = &a.era {
p.generator = GENERATOR_VERSION_V3;
p.era_bytes = Some(eb.clone());
}
println!(
"seed \"{}\" generator v{} attempt {} program id {:016x} seed words {}",
"seed \"{}\" generator v{} class {} attempt {} program id {:016x} seed words {}",
p.seed_string,
p.generator,
p.class.name(),
p.attempt,
p.program_id(),
p.seed.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" ")
);
println!("op mix {} loads/hash {}", p.op_mix(), p.loads_per_hash());
println!("op mix {} loads/hash {} bytes/hash {}", p.op_mix(), p.loads_per_hash(), p.bytes_per_hash());
if let Some(e) = p.class.era {
println!(
"era {} ({}): width {} B, stride mul {:#010x} rot {}, interleave {:?}, windows (site:shrink:offset) {}",
e.label(),
a.era.as_ref().map(|x| x.2.as_str()).unwrap_or("?"),
e.width_words as u32 * 4,
e.stride_mul,
e.stride_rot,
e.pos,
p.instrs
.iter()
.enumerate()
.filter(|(_, i)| i.op == igneum_pow::generator::Op::Load)
.map(|(k, i)| format!("{k}:{}:{}", i.win, i.off))
.collect::<Vec<_>>()
.join(" ")
);
}
for (k, i) in p.instrs.iter().enumerate() {
println!(
"{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}",
"{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}{}",
i.op.name(),
i.dst,
i.src,
@ -254,7 +411,8 @@ fn show(a: &Args) {
i.imm2,
i.rot,
i.bit,
i.mask
i.mask,
if i.op == igneum_pow::generator::Op::Load && i.width > 1 { format!(" width={}B", i.width as u32 * 4) } else { String::new() }
);
}
}

View file

@ -3,12 +3,18 @@
//! ARX-multiply mixer. The verifier holds the cache and never the dataset.
//!
//! All arithmetic is on u32 modulo 2^32. Rotations are by 1..31 at every call site.
//!
//! Counter ASIC 2.0 (5 October 2026, `docs/plans/mixer-x4.md`, behind the program class): the construction has a
//! [`Shape`], the mixer multiplier `m` and the cache size. Under `m` every mixer application of an item becomes
//! `m` applications with distinct round keys, the 8 dependent cache reads unchanged; the cache doubles when the
//! dataset doubles ([`growth_doublings`]). [`Shape::V2`] (`m = 1`, 2^26 words) is version 2 bit for bit.
use crate::generator::LoadClass;
use crate::seed::{day_key, fnv1a64_words, SplitMix64};
pub const CACHE_LOG2_WORDS: usize = 26;
pub const CACHE_SEGMENT_LOG2_LINES: usize = 6;
/// 2^26 words = 256 MiB.
/// 2^26 words = 256 MiB (the version 2 cache, and the v3 cache until the first dataset doubling).
pub const CACHE_WORDS: usize = 1 << CACHE_LOG2_WORDS;
/// 2^22 lines of 16 words.
pub const CACHE_LINES: usize = CACHE_WORDS >> 4;
@ -23,6 +29,102 @@ pub const CHACHA_ROUNDS: usize = 12;
pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574];
/// "Igne", "umMH".
pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48];
/// "Igne", "umHT": the chain tag of the hot table (hot-table experiment, `docs/plans/hot-table.md`).
pub const HOT_TAG: [u32; 2] = [0x49676e65, 0x756d4854];
/// Domain tag of the hot key: `KH = seed_words_from_bytes("igneum-hot/" || epoch seed bytes)`.
pub const HOT_KEY_TAG: &[u8] = b"igneum-hot/";
/// Words per MiB of hot table.
pub const HOT_WORDS_PER_MIB: u32 = 1 << 18;
/// Segments (64 chained lines of 16 words, 4 KiB) per MiB of hot table.
pub const HOT_SEGMENTS_PER_MIB: u32 = 256;
/// The shape of the item derivation and of the cache: the mixer multiplier and the cache size.
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub struct Shape {
/// Mixer applications per round (and after the last read): 1 under version 2, 4 under class v3.
pub mixer_mult: u32,
/// The cache is 2^cache_log2_words words (26 at genesis; 27 and 28 after the dataset doublings of 1.13.3).
pub cache_log2_words: u32,
}
impl Shape {
/// Version 2: one mixer application per round, a 2^26-word cache.
pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32 };
/// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule).
pub fn for_class(class: &LoadClass) -> Shape {
Shape::for_class_day(class, 0)
}
/// The shape of a load class on day `days_since_genesis` of the chain: the class's multiplier, and the cache
/// of [`cache_log2_words`] when the class has the growth rule, else 2^26 words.
pub fn for_class_day(class: &LoadClass, days_since_genesis: u64) -> Shape {
Shape {
mixer_mult: class.mixer_mult(),
cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 },
}
}
pub fn is_v2(&self) -> bool {
*self == Shape::V2
}
pub fn cache_words(&self) -> usize {
1usize << self.cache_log2_words
}
pub fn cache_lines(&self) -> usize {
self.cache_words() >> 4
}
pub fn cache_line_mask(&self) -> u32 {
(self.cache_lines() - 1) as u32
}
pub fn cache_segments(&self) -> usize {
self.cache_lines() >> CACHE_SEGMENT_LOG2_LINES
}
pub fn log2_segments(&self) -> u32 {
self.cache_log2_words - 4 - CACHE_SEGMENT_LOG2_LINES as u32
}
/// Mixer applications per item: `(ITEM_ROUNDS + 1) x m`.
pub fn mixers_per_item(&self) -> u32 {
(ITEM_ROUNDS as u32 + 1) * self.mixer_mult
}
}
// --------------------------------------------------------------------------------------------------------------
// Dataset growth, option C (spec 01 section 1.13.3 option (b) with the cache tied to the dataset's doublings)
// --------------------------------------------------------------------------------------------------------------
/// Days per year of the growth schedule: one year = 31,536,000 DAA seconds of 86,400 (spec 01 section 1.13.3).
pub const GROWTH_DAYS_PER_YEAR: u64 = 365;
/// The linear schedule of 1.13.3, 2 GiB at genesis plus 0.5 GiB per year, is `G x (1 + d / 1460)` for the genesis
/// size `G` and the day `d`: it doubles at day 1,460 (year 4), quadruples at day 4,380 (year 12), reaches 8x at
/// day 10,220 (year 28) and 16x at day 21,900 (year 60).
pub const GROWTH_DOUBLING_DAYS: u64 = 4 * GROWTH_DAYS_PER_YEAR;
/// The number of dataset doublings reached by day `days_since_genesis` of the chain: `floor(log2(1 + d / 1460))`,
/// in integers (`1 + d / 1460` rounded down, then its integer log2, which equals the real log2's floor because a
/// power of two is an integer). 0 until day 1,459; 1 from day 1,460 (year 4); 2 from day 4,380 (year 12).
pub fn growth_doublings(days_since_genesis: u64) -> u32 {
(1 + days_since_genesis / GROWTH_DOUBLING_DAYS).ilog2()
}
/// The cache size on day `d` under option C: 2^26 words doubled once per dataset doubling (256 MiB, 512 MiB from
/// year 4, 1 GiB from year 12).
pub fn cache_log2_words(days_since_genesis: u64) -> u32 {
CACHE_LOG2_WORDS as u32 + growth_doublings(days_since_genesis)
}
/// The dataset size on day `d` under option (b) of 1.13.3: the genesis size (2^`genesis_log2_words` words: 28 for
/// the 1 GiB packs and the devnet, 29 for the designed 2 GiB) doubled once per doubling of the linear schedule. The
/// result is capped at 32 (the item index is 32 bits, spec 1.13.3).
pub fn dataset_log2_words(genesis_log2_words: u32, days_since_genesis: u64) -> u32 {
(genesis_log2_words + growth_doublings(days_since_genesis)).min(32)
}
/// Days since genesis from two day indices of `bind::day_index` (the header's `timestamp_ms / 86,400,000`): the
/// day of the block and the day of the genesis header. A block before the genesis day (clock skew) is day 0.
pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 {
day_index.saturating_sub(genesis_day_index)
}
#[inline(always)]
fn rotl(x: u32, n: u32) -> u32 {
@ -66,17 +168,23 @@ pub fn chacha_block(x: &[u32; 16]) -> [u32; 16] {
y
}
/// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15].
/// Mixer parameters drawn from the day key, plus the [`Shape`] the mixer is applied under. Draw order:
/// ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15]. The shape is not drawn: it is the class's.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct MixParams {
pub key: [u32; 8],
pub rot: [u32; 8],
pub mul: [u32; 16],
pub rc: [u32; 16],
pub shape: Shape,
}
impl MixParams {
/// Version 2 shape.
pub fn new(key: [u32; 8]) -> Self {
Self::with_shape(key, Shape::V2)
}
pub fn with_shape(key: [u32; 8], shape: Shape) -> Self {
let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32));
let mut rot = [0u32; 8];
let mut mul = [0u32; 16];
@ -90,7 +198,7 @@ impl MixParams {
for c in rc.iter_mut() {
*c = rng.next() as u32;
}
Self { key, rot, mul, rc }
Self { key, rot, mul, rc, shape }
}
/// Parameters for a day string: the key is `seed_words("day/" + day)`.
pub fn for_day(day: &str) -> Self {
@ -98,12 +206,84 @@ impl MixParams {
}
}
/// The dataset layout (era layout, `docs/plans/era-layout.md` section 1.2): word `w` of the dataset holds word
/// `j(w)` of item `t(w)`, where `j(w)` gathers the four bits of `w` at the ascending positions `pos` and `t(w)` is
/// `w` with those bits removed. [`Layout::LINEAR`] (`pos = [0, 1, 2, 3]`) is `dataset[w] = item(w >> 4)[w & 15]`,
/// the lottery hash's mapping. Every position is below 16, so the mapping is the same at every dataset size of
/// at least 2^16 words and an item keeps its value at every size.
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub struct Layout {
pub pos: [u8; 4],
}
impl Layout {
pub const LINEAR: Layout = Layout { pos: [0, 1, 2, 3] };
pub fn is_linear(&self) -> bool {
self.pos == [0, 1, 2, 3]
}
/// Positions ascending, distinct, below 16.
pub fn is_valid(&self) -> bool {
self.pos.iter().all(|&p| p < 16) && (1..4).all(|i| self.pos[i] > self.pos[i - 1])
}
/// `(t, j)` of word index `w`.
#[inline(always)]
pub fn split(&self, w: u32) -> (u32, u32) {
if self.is_linear() {
return (w >> 4, w & 15);
}
let mut j = 0u32;
for (i, &p) in self.pos.iter().enumerate() {
j |= ((w >> p) & 1) << i;
}
// remove the highest position first so the lower ones stay where they are
let mut t = w;
for &p in self.pos.iter().rev() {
let p = p as u32;
let low = (1u32 << p) - 1;
t = (t & low) | ((t >> (p + 1)) << p);
}
(t, j)
}
/// The word index of word `j` of item `t`: the inverse of [`Layout::split`].
#[inline(always)]
pub fn join(&self, t: u32, j: u32) -> u32 {
if self.is_linear() {
return (t << 4) | (j & 15);
}
// insert the lowest position first: every later position counts the bit just inserted
let mut w = t;
for (i, &p) in self.pos.iter().enumerate() {
let p = p as u32;
let low = (1u32 << p) - 1;
w = ((w >> p) << (p + 1)) | (w & low) | (((j >> i) & 1) << p);
}
w
}
}
impl Default for Layout {
fn default() -> Self {
Layout::LINEAR
}
}
/// Round key `(r + 1) * 0x9E3779B9` mod 2^32.
#[inline(always)]
pub fn round_key(r: usize) -> u32 {
((r + 1) as u32).wrapping_mul(0x9E3779B9)
}
/// The round key of application `j` (0 <= j < m) of round `r` under multiplier `m`: `round_key(r * m + j)`. For
/// `m = 1` this is `round_key(r)`, version 2's key.
#[inline(always)]
pub fn round_key_mult(r: usize, j: usize, m: usize) -> u32 {
round_key(r * m + j)
}
/// `M_r` on 16 words in place: per word `(s ^ (RC + rk)) * MUL`, then one ChaCha-shaped double round with
/// the four column rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`.
#[inline(always)]
@ -122,9 +302,11 @@ pub fn mixer(s: &mut [u32; 16], rk: u32, mp: &MixParams) {
qr(s, 3, 4, 9, 14, r[4], r[5], r[6], r[7]);
}
/// The 256 MiB cache for one day key.
/// The cache for one day key: 2^log2_words words (256 MiB under version 2).
pub struct Cache {
pub key: [u32; 8],
pub log2_words: u32,
line_mask: u32,
words: Vec<u32>,
}
@ -132,6 +314,12 @@ impl Cache {
/// One segment: 64 chained lines written at `cache[seg * 1024 ..]`.
/// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`.
pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) {
Self::fill_segment_tagged(words, seg, key, &CACHE_TAG)
}
/// [`Cache::fill_segment`] with an explicit chain tag: [`CACHE_TAG`] for the cache, [`HOT_TAG`] for the hot
/// table of `docs/plans/hot-table.md` (the same chain, another key and tag).
pub fn fill_segment_tagged(words: &mut [u32], seg: usize, key: &[u32; 8], tag: &[u32; 2]) {
let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16;
let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16];
let mut prev = [0u32; 16];
@ -141,8 +329,8 @@ impl Cache {
x[4..12].copy_from_slice(key);
x[12] = seg as u32;
x[13] = j as u32;
x[14] = CACHE_TAG[0];
x[15] = CACHE_TAG[1];
x[14] = tag[0];
x[15] = tag[1];
for i in 0..16 {
x[i] ^= prev[i];
}
@ -152,13 +340,22 @@ impl Cache {
}
}
/// The whole cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order.
/// The version 2 cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order.
pub fn fill(key: [u32; 8]) -> Cache {
let mut words = vec![0u32; CACHE_WORDS];
for seg in 0..CACHE_SEGMENTS {
Self::fill_log2(key, CACHE_LOG2_WORDS as u32)
}
/// A cache of 2^`log2_words` words (26, 27 or 28 under the growth rule; smaller sizes for tests): 2^(log2 - 10)
/// independent chains of 64 lines, the same chain function at every size, so a larger cache's first segments
/// are the smaller cache's segments word for word.
pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache {
assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30");
let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words };
let mut words = vec![0u32; shape.cache_words()];
for seg in 0..shape.cache_segments() {
Self::fill_segment(&mut words, seg, &key);
}
Cache { key, words }
Cache { key, log2_words, line_mask: shape.cache_line_mask(), words }
}
pub fn for_day(day: &str) -> Cache {
@ -170,10 +367,27 @@ impl Cache {
&self.words
}
/// Cache line `a` (0 <= a < 2^22) as 16 words.
pub fn lines(&self) -> usize {
self.words.len() >> 4
}
pub fn line_mask(&self) -> u32 {
self.line_mask
}
pub fn segments(&self) -> usize {
self.lines() >> CACHE_SEGMENT_LOG2_LINES
}
/// Cache line `a` (masked to the cache's lines) as 16 words.
#[inline(always)]
pub fn line(&self, a: u32) -> &[u32] {
let o = (a & CACHE_LINE_MASK) as usize * 16;
let o = (a & self.line_mask) as usize * 16;
&self.words[o..o + 16]
}
/// [`Cache::line`] with the mask as a constant (the verifier's hot path, see [`derive_items`]).
#[inline(always)]
pub fn line_const<const LINE_MASK: u32>(&self, a: u32) -> &[u32] {
let o = (a & LINE_MASK) as usize * 16;
&self.words[o..o + 16]
}
@ -183,11 +397,114 @@ impl Cache {
}
}
/// The hot key of an epoch: `seed_words_from_bytes("igneum-hot/" || seed_bytes)`, `seed_bytes` the program seed
/// bytes before any attempt suffix, so every attempt of one epoch shares one table.
pub fn hot_key(seed_bytes: &[u8]) -> [u32; 8] {
let mut b = Vec::with_capacity(HOT_KEY_TAG.len() + seed_bytes.len());
b.extend_from_slice(HOT_KEY_TAG);
b.extend_from_slice(seed_bytes);
crate::seed::seed_words_from_bytes(&b)
}
/// Words of a hot table of `mb` MiB.
pub fn hot_words(mb: u32) -> u32 {
mb * HOT_WORDS_PER_MIB
}
/// Segments of a hot table of `mb` MiB.
pub fn hot_segments(mb: u32) -> u32 {
mb * HOT_SEGMENTS_PER_MIB
}
/// The hot index of a source word: `mulhi(src, words)`, the high 32 bits of the 64-bit product, in `[0, words)`
/// for any table size (the multiply-shift range reduction of spec 01 section 1.13.3).
#[inline(always)]
pub fn hot_index(src: u32, words: u32) -> u32 {
((src as u64 * words as u64) >> 32) as u32
}
/// The hot table `H` of one epoch (hot-table experiment): `mb` MiB of chained ChaCha12 lines under the hot key,
/// read by the hot load slots as `dst ^= H[hot_index(src, words)]`. The verifier holds it beside the cache.
pub struct HotTable {
pub key: [u32; 8],
pub mb: u32,
words: Vec<u32>,
}
impl HotTable {
/// Fill `mb` MiB under `key` on the calling thread.
pub fn fill(key: [u32; 8], mb: u32) -> HotTable {
assert!(mb >= 1 && mb <= 4096, "hot table size in MiB out of range");
let n = hot_words(mb) as usize;
let mut words = vec![0u32; n];
for seg in 0..hot_segments(mb) as usize {
Cache::fill_segment_tagged(&mut words, seg, &key, &HOT_TAG);
}
HotTable { key, mb, words }
}
/// The table of the epoch whose program seed bytes are `seed_bytes`.
pub fn for_seed_bytes(seed_bytes: &[u8], mb: u32) -> HotTable {
Self::fill(hot_key(seed_bytes), mb)
}
#[inline(always)]
pub fn n_words(&self) -> u32 {
self.words.len() as u32
}
/// `H[i]`.
#[inline(always)]
pub fn at(&self, i: u32) -> u32 {
self.words[i as usize]
}
/// `H[hot_index(src, words)]`: what a hot load reads for source word `src`.
#[inline(always)]
pub fn word(&self, src: u32) -> u32 {
self.words[hot_index(src, self.n_words()) as usize]
}
#[inline(always)]
pub fn words(&self) -> &[u32] {
&self.words
}
/// FNV-1a 64 over the table as little-endian bytes (what `vectors.h` carries as `IGNEUM_HOT_FNV64`).
pub fn fnv1a64(&self) -> u64 {
fnv1a64_words(&self.words)
}
}
/// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of
/// independent items overlap in the memory system (`deriveItems` in the Swift).
/// independent items overlap in the memory system (`deriveItems` in the Swift). Under multiplier `m`
/// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its
/// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2.
pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
// The item loop lives in its own function, one instance per cache size the growth rule can reach with the line
// mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit
// against 0.61 out of line (the version 2 verifier, bisected on one core under the measure lock, 5 October 2026,
// `docs/plans/mixer-x4.md` section 6.6: the constant mask alone, or the mask hoisted into a local, or the
// constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any
// other cache size (tests) takes the instance with the run-time mask.
match cache.log2_words {
26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, out),
27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, out),
28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, out),
29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, out),
30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, out),
_ => derive_items_mask::<0>(ts, mp, cache, out),
}
}
/// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept
/// out of line on purpose (see [`derive_items`]).
#[inline(never)]
fn derive_items_mask<const LINE_MASK: u32>(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
let n = ts.len();
debug_assert!(out.len() >= n);
debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask);
let m = mp.shape.mixer_mult as usize;
for k in 0..n {
let s = &mut out[k];
let t = ts[k];
@ -197,20 +514,24 @@ pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32;
}
}
for r in 0..ITEM_ROUNDS {
let rk = round_key(r);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
for j in 0..m {
let rk = round_key_mult(r, j, m);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
for s in out[..n].iter_mut() {
let line = cache.line(s[0]);
let line = if LINE_MASK != 0 { cache.line_const::<LINE_MASK>(s[0]) } else { cache.line(s[0]) };
for i in 0..16 {
s[i] ^= line[i];
}
}
}
let rk = round_key(ITEM_ROUNDS);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
for j in 0..m {
let rk = round_key_mult(ITEM_ROUNDS, j, m);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
}
@ -221,7 +542,7 @@ pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
out[0]
}
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters and the 256 MiB cache.
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape) and the cache.
pub struct MemhardCpu {
pub params: MixParams,
pub cache: Cache,
@ -231,26 +552,41 @@ pub struct MemhardCpu {
pub const FETCH_MAX: usize = 64;
impl MemhardCpu {
/// Version 2 shape.
pub fn new(key: [u32; 8]) -> Self {
Self { params: MixParams::new(key), cache: Cache::fill(key) }
Self::with_shape(key, Shape::V2)
}
pub fn with_shape(key: [u32; 8], shape: Shape) -> Self {
Self { params: MixParams::with_shape(key, shape), cache: Cache::fill_log2(key, shape.cache_log2_words) }
}
pub fn for_day(day: &str) -> Self {
Self::new(day_key(day))
}
/// `dataset[w] = item(w >> 4)[w & 15]`.
pub fn shape(&self) -> Shape {
self.params.shape
}
/// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout).
pub fn word(&self, w: u32) -> u32 {
derive_item(w >> 4, &self.params, &self.cache)[(w & 15) as usize]
self.word_at(Layout::LINEAR, w)
}
/// `dataset[w] = item(t(w))[j(w)]` under `layout` (era layout; the layout is the program's, the cache the
/// day's, so one cache serves every era of a day).
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
let (t, j) = layout.split(w);
derive_item(t, &self.params, &self.cache)[j as usize]
}
/// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once.
/// Returns the number of distinct items derived.
pub fn fetch(&self, idx: &[u32], out: &mut [u32]) -> usize {
pub fn fetch(&self, idx: &[u32], out: &mut [u32], layout: Layout) -> usize {
let n = idx.len();
assert!(n <= FETCH_MAX && out.len() >= n);
let mut uniq = [0u32; FETCH_MAX];
let mut slot = [0u8; FETCH_MAX];
let mut word = [0u8; FETCH_MAX];
let mut u = 0usize;
for k in 0..n {
let t = idx[k] >> 4;
let (t, j) = layout.split(idx[k]);
word[k] = j as u8;
let found = uniq[..u].iter().position(|&x| x == t);
let j = match found {
Some(j) => j,
@ -265,7 +601,39 @@ impl MemhardCpu {
let mut items = [[0u32; 16]; FETCH_MAX];
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
for k in 0..n {
out[k] = items[slot[k] as usize][(idx[k] & 15) as usize];
out[k] = items[slot[k] as usize][word[k] as usize];
}
u
}
/// `out[k][j] = dataset[base[k] + j]` for `j < width` (read-width experiment): `base[k]` is aligned to `width`
/// words and the layout's low `log2(width)` positions are the identity, so every lane's words lie in one item
/// at consecutive word offsets, derived once per distinct item. Returns the distinct items.
pub fn fetch_wide(&self, base: &[u32], width: usize, out: &mut [[u32; 16]], layout: Layout) -> usize {
let n = base.len();
assert!(n <= FETCH_MAX && out.len() >= n && width <= 16);
debug_assert!((0..width.trailing_zeros() as usize).all(|i| layout.pos[i] == i as u8), "a wide load needs the identity on its low positions");
let mut uniq = [0u32; FETCH_MAX];
let mut slot = [0u8; FETCH_MAX];
let mut word = [0u8; FETCH_MAX];
let mut u = 0usize;
for k in 0..n {
let (t, j0) = layout.split(base[k]);
word[k] = j0 as u8;
let j = match uniq[..u].iter().position(|&x| x == t) {
Some(j) => j,
None => {
uniq[u] = t;
u += 1;
u - 1
}
};
slot[k] = j as u8;
}
let mut items = [[0u32; 16]; FETCH_MAX];
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
for k in 0..n {
let o = word[k] as usize;
out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]);
}
u
}
@ -285,6 +653,40 @@ mod tests {
assert_eq!(mp.rc[0], 0xbab68293);
assert_eq!(mp.rc[15], 0x31b49ee2);
assert!(mp.mul.iter().all(|m| m & 1 == 1));
assert_eq!(mp.shape, Shape::V2);
}
/// Era layout: split and join are inverse, the linear layout is today's mapping, and an interleaved layout
/// keeps every position below 16 so the mapping is the same at every size of at least 2^16 words.
#[test]
fn layout_split_join() {
let lin = Layout::LINEAR;
assert!(lin.is_linear() && lin.is_valid());
for w in [0u32, 1, 15, 16, 17, 0x0fff_ffff, 0xffff_ffff] {
assert_eq!(lin.split(w), (w >> 4, w & 15));
assert_eq!(lin.join(w >> 4, w & 15), w);
}
let l = Layout { pos: [0, 1, 7, 12] };
assert!(!l.is_linear() && l.is_valid());
for w in [0u32, 1, 2, 3, 4, 127, 128, 129, 4095, 4096, 0x0fff_ffff, 0x1234_5678, 0xffff_ffff] {
let (t, j) = l.split(w);
assert!(j < 16);
assert_eq!(l.join(t, j), w, "w {w:#x}");
}
// bits: j0 = bit 0, j1 = bit 1, j2 = bit 7, j3 = bit 12; t = the other 28 bits in order
assert_eq!(l.split(0b1_0000_0000_0000), (0, 8));
assert_eq!(l.split(1 << 7), (0, 4));
assert_eq!(l.split(0b100), (1, 0));
// every t in 0..2^(D-4) appears exactly once among w < 2^D (D = 16), with every j
let mut seen = vec![0u32; 1 << 12];
for w in 0..(1u32 << 16) {
let (t, j) = l.split(w);
seen[t as usize] |= 1 << j;
}
assert!(seen.iter().all(|&s| s == 0xffff));
assert!(!Layout { pos: [0, 1, 1, 5] }.is_valid());
assert!(!Layout { pos: [0, 1, 2, 16] }.is_valid());
assert!(!Layout { pos: [1, 0, 2, 3] }.is_valid());
}
#[test]
@ -296,6 +698,40 @@ mod tests {
assert_eq!(y, z);
}
/// Hot-table experiment: the genesis epoch's table (seed bytes "igneum-genesis") as the hot packs carry it
/// (`proto-cuda/packs-ca2-hot/hot32k4/vectors.json`: hot_head, hot_fnv1a64; the head is the same at every size,
/// a larger table is more segments). The index mapping stays inside the table for any size.
#[test]
fn hot_table_fill_vector_and_index() {
assert_ne!(HOT_TAG, CACHE_TAG);
let h = HotTable::for_seed_bytes(b"igneum-genesis", 32);
assert_eq!(h.n_words(), 1 << 23);
assert_eq!(hot_segments(32), 8192);
assert_eq!(
&h.words()[..16],
&[
0x8068cc73, 0x6036ebf9, 0xb604cd25, 0x8ffb840e, 0xc54074a2, 0x285c0695, 0x77512425, 0xc26a58a7,
0x72c88757, 0xc10fca78, 0x513825dd, 0x30d6ccc8, 0x9a05e7cf, 0xb9533f50, 0x4bac3ba0, 0xa5c19528
]
);
assert_eq!(h.fnv1a64(), 0xc1767ba3ef02719f, "hot32k4 pack, hot_fnv1a64");
assert_eq!(h.key, hot_key(b"igneum-genesis"));
assert_ne!(h.key, day_key("2026-10-03"));
// a different seed, a different table; the same seed under the cache tag is not the hot table
assert_ne!(HotTable::for_seed_bytes(b"igneum-genesis\x01\x00\x00\x00", 1).words()[..16], h.words()[..16]);
let mut under_cache_tag = vec![0u32; 1024];
Cache::fill_segment(&mut under_cache_tag, 0, &h.key);
assert_ne!(&under_cache_tag[..16], &h.words()[..16]);
for words in [hot_words(32), hot_words(64), hot_words(96)] {
assert_eq!(hot_index(0, words), 0);
assert!(hot_index(u32::MAX, words) < words);
assert_eq!(hot_index(u32::MAX, words), words - 1);
assert!(hot_index(0x8000_0000, words) == words / 2);
}
assert_eq!(hot_index(0x1234_5678, 1 << 24), 0x1234_5678 >> 8);
assert_eq!(h.word(0x8000_0000), h.at(1 << 22));
}
#[test]
fn first_cache_line_matches_pack() {
// vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0.
@ -310,4 +746,96 @@ mod tests {
]
);
}
/// Option C: the schedule table of `docs/plans/mixer-x4.md` (day -> doublings, cache words, dataset words at a
/// 2^28 genesis). The doublings fall at years 4 and 12 exactly, never a day early.
#[test]
fn growth_schedule_table() {
let table: [(u64, u32, u32, u32); 12] = [
(0, 0, 26, 28),
(1, 0, 26, 28),
(365, 0, 26, 28),
(1_459, 0, 26, 28),
(1_460, 1, 27, 29),
(2_920, 1, 27, 29),
(4_379, 1, 27, 29),
(4_380, 2, 28, 30),
(10_219, 2, 28, 30),
(10_220, 3, 29, 31),
(21_900, 4, 30, 32),
(100_000, 6, 32, 32),
];
for (d, k, c, s) in table {
assert_eq!(growth_doublings(d), k, "day {d}");
assert_eq!(cache_log2_words(d), c, "day {d}");
assert_eq!(dataset_log2_words(28, d), s, "day {d}");
}
// the designed 2 GiB genesis: 2^29 words, 2^30 at year 4, 2^31 at year 12
assert_eq!(dataset_log2_words(29, 0), 29);
assert_eq!(dataset_log2_words(29, 1_460), 30);
assert_eq!(dataset_log2_words(29, 4_380), 31);
// the linear schedule itself: 2 GiB x (1 + d / 1460) crosses 4 GiB at day 1,460 and 8 GiB at day 4,380
for d in [1_459u64, 1_460, 4_379, 4_380] {
let bytes = 2u64 * (1 << 30) + (1u64 << 29) * d / 365;
let k = (bytes / (2u64 << 30)).ilog2();
assert_eq!(growth_doublings(d), k, "day {d}: linear {bytes} bytes");
}
assert_eq!(days_since_genesis(20_730, 20_729), 1);
assert_eq!(days_since_genesis(20_729, 20_729), 0);
assert_eq!(days_since_genesis(20_000, 20_729), 0);
let v2 = Shape::for_class_day(&LoadClass::V2, 100_000);
assert_eq!(v2, Shape::V2);
let v3 = Shape::for_class_day(&LoadClass::MX4, 0);
assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26 });
assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27);
assert_eq!(v3.mixers_per_item(), 36);
assert_eq!(Shape::V2.mixers_per_item(), 9);
assert_eq!(Shape::V2.cache_segments(), CACHE_SEGMENTS);
assert_eq!(Shape::V2.cache_line_mask(), CACHE_LINE_MASK);
assert_eq!(Shape::V2.log2_segments(), 16);
}
/// The multiplied mixer, restated by hand on a small cache: `m` applications with keys `round_key(r m + j)`
/// before every read, the same 8 reads; `m = 1` is `derive_item` of version 2 word for word; a larger cache's
/// first segments equal the smaller cache's.
#[test]
fn mixer_mult_by_hand() {
let key = day_key("2026-10-03");
let small = Cache::fill_log2(key, 16);
let big = Cache::fill_log2(key, 18);
assert_eq!(&big.words()[..small.words().len()], small.words());
assert_eq!(small.segments(), 64);
assert_eq!(small.line_mask(), 4095);
for m in [1u32, 2, 4] {
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16 });
for t in [0u32, 1, 12_345, u32::MAX] {
let got = derive_item(t, &mp, &small);
let mut s = [0u32; 16];
s[..8].copy_from_slice(&key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
for r in 0..8usize {
for j in 0..m as usize {
mixer(&mut s, round_key(r * m as usize + j), &mp);
}
let line = small.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
for j in 0..m as usize {
mixer(&mut s, round_key(8 * m as usize + j), &mp);
}
assert_eq!(got, s, "m {m} t {t}");
}
}
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16 });
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16 });
assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small));
assert_eq!(round_key_mult(0, 0, 1), round_key(0));
assert_eq!(round_key_mult(8, 0, 1), round_key(8));
assert_eq!(round_key_mult(2, 3, 4), round_key(11));
assert_eq!(round_key_mult(8, 3, 4), round_key(35));
}
}

View file

@ -10,7 +10,7 @@
//! a later attempt. This module is the one place that rule is written in Rust; the miner checks every pack it
//! writes with it before a worker sees the pack, and the tests pin the attempt vectors the C side also pins.
use crate::generator::attempt_words;
use crate::generator::{attempt_words, ProgramClass};
use crate::seed::seed_words_from_bytes;
use std::fmt;
use std::path::Path;
@ -23,6 +23,12 @@ pub struct PackIdentity {
pub attempt: u32,
pub seedw: [u32; 8],
pub keyw: [u32; 8],
/// `IGNEUM_GENERATOR` (2 or 3; a pack without the line is generator 1, which no worker runs).
pub generator: u32,
/// The program class the generator version names (Counter ASIC 2.0).
pub class: ProgramClass,
/// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs).
pub era_hex: Option<String>,
}
/// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry.
@ -35,6 +41,9 @@ pub enum PackFault {
/// The files of one pack contradict each other (seeds.txt against program.h, or the init words against the
/// seeds and the attempt): a half-written or hand-edited pack, or a worker and an exporter on different rules.
Disagree(String),
/// The pack is of another program class than the one the chain is on (spec 01 section 1.4.5: an implementation
/// refuses a pack whose generator version is not its own), or its era seed is not the era the job names.
WrongClass(String),
}
impl fmt::Display for PackFault {
@ -50,6 +59,7 @@ impl fmt::Display for PackFault {
day_label(want_day)
),
PackFault::Disagree(w) => write!(f, "program pack and its seeds disagree: {w}"),
PackFault::WrongClass(w) => write!(f, "program pack of the wrong class: {w}"),
}
}
}
@ -151,6 +161,44 @@ fn seeds_line(text: &str, key: &str) -> Option<String> {
/// Checks the texts of a pack (program.h, and seeds.txt when it exists) against the seeds a worker will be asked
/// to mine with. Pure: the miner and the tests call it with file contents.
pub fn verify_pack_texts(program_h: &str, seeds_txt: Option<&str>, want_epoch: &[u8], want_day: &[u8]) -> Result<PackIdentity, PackFault> {
verify_pack_texts_chain(program_h, seeds_txt, want_epoch, want_day, None, None)
}
/// [`verify_pack_texts`] that also demands a program class and, for class v3, the era seed the chain is on
/// (Counter ASIC 2.0, 5 October 2026). `want_class` `None` accepts either class; `want_era` `None` skips the era.
/// A pack whose `IGNEUM_GENERATOR` is neither 2 nor 3 is refused whatever is wanted.
pub fn verify_pack_texts_chain(
program_h: &str,
seeds_txt: Option<&str>,
want_epoch: &[u8],
want_day: &[u8],
want_class: Option<ProgramClass>,
want_era: Option<&[u8]>,
) -> Result<PackIdentity, PackFault> {
let generator = define_u32(program_h, "IGNEUM_GENERATOR").unwrap_or(1);
let Some(class) = ProgramClass::from_generator(generator) else {
return Err(PackFault::WrongClass(format!("IGNEUM_GENERATOR {generator} is not a generator version this software runs (2 or 3)")));
};
// IGNEUM_PROGRAM_CLASS, when present, must name the class the generator version names
if let Some(named) = define_str(program_h, "IGNEUM_PROGRAM_CLASS") {
if ProgramClass::parse(&named) != Some(class) {
return Err(PackFault::Disagree(format!("IGNEUM_PROGRAM_CLASS {named:?} does not match IGNEUM_GENERATOR {generator}")));
}
}
let era_hex = define_str(program_h, "IGNEUM_ERA_SEED_HEX").map(|h| h.to_ascii_lowercase());
if let Some(want) = want_class {
if want != class {
return Err(PackFault::WrongClass(format!("the pack is program class {} (generator {generator}), the chain is on class {}", class.name(), want.name())));
}
}
if let (Some(want), ProgramClass::V3) = (want_era, class) {
let want_hex = hex(want);
match &era_hex {
Some(h) if *h == want_hex => {}
Some(h) => return Err(PackFault::WrongClass(format!("the pack's era seed {} is not the era seed {} the job names", short(h), short(&want_hex)))),
None => return Err(PackFault::WrongClass("a class v3 pack without IGNEUM_ERA_SEED_HEX; the job names an era seed".into())),
}
}
let seedw = define_words(program_h, "IGNEUM_SEEDW_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_SEEDW_INIT with 8 words".into()))?;
let keyw = define_words(program_h, "IGNEUM_KEY_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_KEY_INIT with 8 words".into()))?;
let attempt = define_u32(program_h, "IGNEUM_PROGRAM_ATTEMPT").unwrap_or(0);
@ -194,14 +242,19 @@ pub fn verify_pack_texts(program_h: &str, seeds_txt: Option<&str>, want_epoch: &
if epoch_hex != want_epoch_hex || day_hex != want_day_hex {
return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex });
}
Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw })
Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex })
}
/// [`verify_pack_texts`] over a pack directory.
pub fn verify_pack_dir(dir: &Path, want_epoch: &[u8], want_day: &[u8]) -> Result<PackIdentity, PackFault> {
verify_pack_dir_chain(dir, want_epoch, want_day, None, None)
}
/// [`verify_pack_texts_chain`] over a pack directory.
pub fn verify_pack_dir_chain(dir: &Path, want_epoch: &[u8], want_day: &[u8], want_class: Option<ProgramClass>, want_era: Option<&[u8]>) -> Result<PackIdentity, PackFault> {
let program_h = std::fs::read_to_string(dir.join("program.h")).map_err(|e| PackFault::Unreadable(format!("cannot read {}/program.h: {e}", dir.display())))?;
let seeds = std::fs::read_to_string(dir.join("seeds.txt")).ok();
verify_pack_texts(&program_h, seeds.as_deref(), want_epoch, want_day)
verify_pack_texts_chain(&program_h, seeds.as_deref(), want_epoch, want_day, want_class, want_era)
}
#[cfg(test)]
@ -301,4 +354,44 @@ mod tests {
assert_eq!(day_label(DAY_20731), "20731");
assert_eq!(day_label("abcd"), "abcd");
}
/// Counter ASIC 2.0: a class v3 chain pack carries generator 3, the class line and the era seed; it is refused
/// when the chain wants class v2, when the era differs, and a v2 pack is refused when the chain wants v3; a
/// generator this software does not run is refused whatever is wanted.
#[test]
fn program_class_and_era_are_checked() {
let e = bytes(EPOCH_34);
let d = bytes(DAY_20731);
let era = [0x5au8; 32];
let v3 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V3, "class test");
let h3 = program_header(&v3.program, "test day", &v3.dataset);
assert!(h3.contains("#define IGNEUM_GENERATOR 3\n"));
assert!(h3.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
assert!(h3.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era))));
let id = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (3, ProgramClass::V3, Some(hex(&era).as_str())));
assert_eq!(id.attempt, v3.program.attempt);
assert!(verify_pack_texts(&h3, None, &e, &d).is_ok(), "no class wanted: either class passes");
let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V2), None).unwrap_err();
assert!(matches!(err, PackFault::WrongClass(_)), "{err}");
assert!(err.to_string().contains("program pack of the wrong class"), "{err}");
let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&[1u8; 32])).unwrap_err();
assert!(err.to_string().contains("era seed"), "{err}");
// the v2 pack of the same seeds: generator 2, no class line, no era line, refused when v3 is wanted
let v2 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V2, "class test");
let h2 = program_header(&v2.program, "test day", &v2.dataset);
assert!(h2.contains("#define IGNEUM_GENERATOR 2\n"));
assert!(!h2.contains("IGNEUM_PROGRAM_CLASS") && !h2.contains("IGNEUM_ERA_SEED_HEX"));
let plain = Epoch::from_seed_bytes(&e, &d, "class test");
assert_eq!(program_header(&plain.program, "test day", &plain.dataset), h2, "class v2 from the chain is the v2 export byte for byte");
let id = verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V2), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex), (2, ProgramClass::V2, None));
assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_))));
// a generator nobody runs
let h9 = h2.replace("#define IGNEUM_GENERATOR 2\n", "#define IGNEUM_GENERATOR 9\n");
assert!(matches!(verify_pack_texts(&h9, None, &e, &d), Err(PackFault::WrongClass(_))));
// a class line that contradicts the generator
let bad = h3.replace("#define IGNEUM_PROGRAM_CLASS \"v3\"\n", "#define IGNEUM_PROGRAM_CLASS \"v2\"\n");
assert!(matches!(verify_pack_texts(&bad, None, &e, &d), Err(PackFault::Disagree(_))));
}
}

View file

@ -1,10 +1,137 @@
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
use crate::generator::{generate, Instr, Op, Program, ITERATIONS, LANES};
use crate::memhard::MemhardCpu;
use crate::generator::{generate, generate_class, EraParams, Instr, LoadClass, Op, Program, ProgramClass, ITERATIONS, LANES};
use crate::memhard::{hot_index, HotTable, Layout, MemhardCpu, Shape};
use crate::seed::day_key;
/// The load address of an era program (`docs/plans/era-layout.md` section 1.3): `y = rotl(x * M, R)`, then the
/// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off &
/// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is
/// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`.
#[inline(always)]
pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 {
match era {
None => x & mask,
Some(e) => {
let (wm, off) = window(ins, mask, log2);
let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot);
((y & wm) | off) & mask
}
}
}
/// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that
/// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words.
#[inline(always)]
pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) {
let k = (ins.win as u32).min(log2.saturating_sub(26));
let wm = mask >> k;
let off = ((ins.off as u32) & ((1u32 << k) - 1)) << (log2 - k);
(wm, off)
}
/// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`:
/// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the
/// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words),
/// so no function of the line alone replaces it: two different lines give two different maps of `dst`, and a
/// dataset of folded lines cannot be stored in place of the dataset (see `docs/plans/read-width.md`).
pub const FOLD_ROT: u32 = 11;
pub const FOLD_MUL: u32 = 0x9E3779B1;
/// The fold of `words` into `dst` (at least one word).
#[inline(always)]
pub fn fold_words(dst: u32, words: &[u32]) -> u32 {
let mut x = dst ^ words[0];
for &w in &words[1..] {
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w;
}
x
}
/// Variant 5 (scratch): the fill value of word `j` (0..2) of slot `slot` of lane `lane` of the unit at base nonce
/// `base`, under program seed words `seed`. The scratch of a unit starts as these values; a slot written during
/// the unit's hash holds what was written. Mirrored as `scr_fill` in every emitted kernel.
#[inline(always)]
pub fn scratch_fill(seed: &[u32; 8], base: u32, lane: u32, slot: u32, j: u32) -> u32 {
splitmix32(
(base.wrapping_add(lane) ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E3779B1))
.wrapping_add((j + 1).wrapping_mul(0x85EBCA77)),
)
}
/// Variant 5: the 16-byte slot after a read-modify-write that read `w` and folded to `x`: `(x ^ w1, rotl(x, 7) ^ w2,
/// x + w0)` behind the slot's tag.
#[inline(always)]
pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
}
/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per
/// resident warp with a per-unit tag per slot.
pub struct ScratchModel {
slots: usize,
written: Vec<bool>,
data: Vec<[u32; 3]>,
pub reads: usize,
pub writes: usize,
/// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every
/// read-modify-write is appended as it happened. `None` on every verification path.
pub trace: Option<Vec<ScratchEvent>>,
}
/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests).
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct ScratchEvent {
pub lane: u8,
pub slot: u32,
/// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill.
pub hit: bool,
pub read: [u32; 3],
/// The fold result, the new value of `dst`.
pub x: u32,
pub written: [u32; 3],
}
impl ScratchModel {
pub fn new(slots_per_lane: usize) -> Self {
Self {
slots: slots_per_lane,
written: vec![false; LANES * slots_per_lane],
data: vec![[0; 3]; LANES * slots_per_lane],
reads: 0,
writes: 0,
trace: None,
}
}
/// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read.
#[inline]
pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 {
let i = lane * self.slots + slot as usize;
let w = if self.written[i] {
self.data[i]
} else {
[
scratch_fill(seed, base, lane as u32, slot, 0),
scratch_fill(seed, base, lane as u32, slot, 1),
scratch_fill(seed, base, lane as u32, slot, 2),
]
};
let x = fold_words(dst, &w);
let out = scratch_rewrite(x, &w);
if let Some(t) = self.trace.as_mut() {
t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out });
}
self.data[i] = out;
self.written[i] = true;
self.reads += 1;
self.writes += 1;
x
}
}
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
#[inline(always)]
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
@ -65,24 +192,55 @@ pub struct DatasetSource {
/// in packs so any implementation can rebuild the key. Empty when the key was given directly.
pub key_bytes: Vec<u8>,
pub dataset: Dataset,
/// The hot table of the epoch (hot-table experiment, `docs/plans/hot-table.md`): `Some` when the program's
/// class has one; filled by [`Epoch::new_class`] and [`Epoch::from_seed_bytes_class`] from the program's seed
/// bytes. A hot load reads `hot[hot_index(src, words)]`.
pub hot: Option<HotTable>,
}
impl DatasetSource {
/// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread.
pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self {
let mut ds = Self::from_key(day_key(day), mode, log2_words);
Self::new_shape(day, mode, log2_words, Shape::V2)
}
/// [`DatasetSource::new`] with the construction's shape (mixer multiplier, cache size; Counter ASIC 2.0).
pub fn new_shape(day: &str, mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
let mut ds = Self::from_key_shape(day_key(day), mode, log2_words, shape);
ds.key_bytes = format!("day/{day}").into_bytes();
ds
}
pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self {
Self::from_key_shape(key, mode, log2_words, Shape::V2)
}
/// [`DatasetSource::from_key`] with the construction's shape. Memory-hard mode fills a cache of
/// `2^shape.cache_log2_words` words on the calling thread.
pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32");
let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 };
let dataset = match mode {
DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] },
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::new(key)),
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)),
};
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset }
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None }
}
/// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB).
pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self {
self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb));
self
}
/// The hot table of a program's class, filled from its seed bytes (none for a class without one).
pub fn attach_hot_for(&mut self, program: &Program) {
self.hot = program.class.hot.map(|h| HotTable::for_seed_bytes(&program.seed_bytes, h.mb as u32));
}
/// The shape of the memory-hard construction ([`Shape::V2`] for the closed form, which has none).
pub fn shape(&self) -> Shape {
self.memhard().map(|m| m.shape()).unwrap_or(Shape::V2)
}
pub fn mode(&self) -> DatasetMode {
@ -99,18 +257,23 @@ impl DatasetSource {
}
}
/// `dataset[w & mask]`.
/// `dataset[w & mask]` under the linear layout (the lottery hash).
pub fn word(&self, w: u32) -> u32 {
self.word_at(Layout::LINEAR, w)
}
/// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it.
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
let w = w & self.mask;
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1),
Dataset::MemoryHard(m) => m.word(w),
Dataset::MemoryHard(m) => m.word_at(layout, w),
}
}
/// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form).
#[inline]
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES]) -> usize {
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES], layout: Layout) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
@ -118,7 +281,24 @@ impl DatasetSource {
}
0
}
Dataset::MemoryHard(m) => m.fetch(idx, out),
Dataset::MemoryHard(m) => m.fetch(idx, out, layout),
}
}
/// `out[k][j] = dataset[base[k] + j]` for `j < width`; bases are masked and aligned to `width` words
/// (`width` 4 or 16, so a lane's words lie in one item). Returns items derived (0 for the closed form).
#[inline]
fn fetch_wide(&self, base: &[u32; LANES], width: usize, out: &mut [[u32; 16]; LANES], layout: Layout) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
for j in 0..width {
out[k][j] = dataset_elem(base[k] + j as u32, *d0, *d1);
}
}
0
}
Dataset::MemoryHard(m) => m.fetch_wide(base, width, out, layout),
}
}
}
@ -146,7 +326,23 @@ pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) ->
/// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`;
/// a block uses `I = bind::block_init_words(H, nonce)`.
pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult {
interpret_warp_scratch(program, seed, base_nonce, ds, false).0
}
/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order
/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for
/// a class without a scratch. For the soundness tests of variant 5 only.
pub fn interpret_warp_scratch(
program: &Program,
seed: &[u32; 8],
base_nonce: u32,
ds: &DatasetSource,
trace: bool,
) -> (WarpResult, Vec<ScratchEvent>) {
let mask = ds.mask;
let log2 = ds.log2_words;
let era = program.class.era;
let layout = program.class.layout();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base_nonce.wrapping_add(lane as u32);
@ -160,10 +356,29 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
let mut items_derived = 0usize;
let mut idx = [0u32; LANES];
let mut val = [0u32; LANES];
let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None };
if trace {
if let Some(m) = scratch.as_mut() {
m.trace = Some(Vec::new());
}
}
let slot_mask = program.class.scratch_slot_mask();
if program.has_hot() {
let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source");
assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's");
}
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &program.instrs {
step(ins, &mut r, &sel, mask, ds, &mut idx, &mut val, &mut items_derived);
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
if ins.op == Op::Scratch {
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
let (d, a) = (ins.dst as usize, ins.src as usize);
for lane in 0..LANES {
let slot = r[a][lane] & slot_mask;
r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]);
}
}
}
}
let mut hashes = [0u64; LANES];
@ -172,7 +387,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
hashes[lane] = ((hi as u64) << 32) | lo as u64;
}
WarpResult { hashes, items_derived }
let events = scratch.and_then(|m| m.trace).unwrap_or_default();
(WarpResult { hashes, items_derived }, events)
}
#[inline(always)]
@ -182,6 +398,9 @@ fn step(
r: &mut [[u32; LANES]; 8],
sel: &[u32; LANES],
mask: u32,
log2: u32,
era: Option<&EraParams>,
layout: Layout,
ds: &DatasetSource,
idx: &mut [u32; LANES],
val: &mut [u32; LANES],
@ -255,22 +474,46 @@ fn step(
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load => {
Op::Load if ins.width == 1 => {
for lane in 0..LANES {
idx[lane] = r[a][lane] & mask;
idx[lane] = load_index(era, ins, r[a][lane], mask, log2);
}
*items_derived += ds.fetch(idx, val);
*items_derived += ds.fetch(idx, val, layout);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
}
Op::Load => {
// Read-width experiment: `width` words from the aligned address, every word folded into dst.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align;
}
let mut vals = [[0u32; 16]; LANES];
*items_derived += ds.fetch_wide(idx, width, &mut vals, layout);
for lane in 0..LANES {
r[d][lane] = fold_words(r[d][lane], &vals[lane][..width]);
}
}
Op::Scratch => {
// handled by the caller (interpret_warp_init), which owns the unit's scratch model
}
Op::Hot => {
// Hot-table experiment: one word of the epoch table at the multiply-shift index, plain xor fold.
let h = ds.hot.as_ref().expect("a hot load needs the hot table");
let n = h.n_words();
for lane in 0..LANES {
r[d][lane] ^= h.at(hot_index(r[a][lane], n));
}
}
Op::WLoad => {
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
let base = (r[a][0] & mask) & !31;
for lane in 0..LANES {
idx[lane] = base + lane as u32;
}
*items_derived += ds.fetch(idx, val);
*items_derived += ds.fetch(idx, val, Layout::LINEAR);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
@ -294,11 +537,39 @@ pub struct Epoch {
/// Default dataset size: 2^28 words = 1 GiB.
pub const DEFAULT_DATASET_LOG2: u32 = 28;
/// Days a day index lies after the network's genesis day (0 for the genesis day and any day before it). The node's
/// entry; the same function as `memhard::days_since_genesis`.
pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 {
crate::memhard::days_since_genesis(day_index, genesis_day_index)
}
impl Epoch {
pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self {
Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) }
}
/// [`Epoch::new`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer multiplier
/// shapes the dataset, the cache is the genesis size since a string day has no day index).
pub fn new_class(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass) -> Self {
Self::new_class_day(seed, day, mode, dataset_log2, class, 0)
}
/// [`Epoch::new_class`] on day `days_since_genesis` of the growth schedule (the cache of
/// `memhard::cache_log2_words` for a class with the growth rule; the dataset size is the caller's).
pub fn new_class_day(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass, days_since_genesis: u64) -> Self {
let shape = Shape::for_class_day(&class, days_since_genesis);
let program = generate_class(seed, class);
let mut dataset = DatasetSource::new_shape(day, mode, dataset_log2, shape);
// hot-table experiment: a hot class fills its table from the seed bytes
dataset.attach_hot_for(&program);
Self { program, dataset }
}
/// `dataset[w]` as this epoch's program reads it: under the program's layout (era layout; linear for v2).
pub fn dataset_word(&self, w: u32) -> u32 {
self.dataset.word_at(self.program.class.layout(), w)
}
/// The production shape: memory-hard, 1 GiB dataset.
pub fn memory_hard(seed: &str, day: &str) -> Self {
Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2)
@ -309,13 +580,71 @@ impl Epoch {
/// `seed_words_from_bytes(day_bytes)` (`bind::day_bytes`). Memory-hard, 1 GiB dataset. `label` is only
/// recorded in emitted packs.
pub fn from_seed_bytes(epoch_seed: &[u8], day_bytes: &[u8], label: &str) -> Self {
let program = crate::generator::generate_from_seed_bytes(label, epoch_seed);
Self::from_seed_bytes_class(epoch_seed, day_bytes, label, LoadClass::V2)
}
/// [`Epoch::from_seed_bytes`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer
/// multiplier shapes the dataset). Day 0 of the growth schedule: the 2^26-word cache and the 2^28-word dataset,
/// which is every devnet pack and vector. A node past the first doubling calls [`Epoch::from_seed_bytes_day`].
pub fn from_seed_bytes_class(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass) -> Self {
Self::from_seed_bytes_day(epoch_seed, day_bytes, label, class, 0, DEFAULT_DATASET_LOG2)
}
/// The chain's shape on day `days_since_genesis` (`memhard::days_since_genesis(day_index(header), day_index(genesis))`,
/// the node's two day indices): the program of the class, and under the class's growth rule the cache of
/// `memhard::cache_log2_words(d)` and the dataset of `memhard::dataset_log2_words(genesis_dataset_log2, d)`
/// (the genesis size is 28 for the 1 GiB devnet, 29 for the designed 2 GiB). Without the growth rule the cache
/// is 2^26 words and the dataset `2^genesis_dataset_log2` on every day.
pub fn from_seed_bytes_day(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> Self {
let program = crate::generator::generate_from_seed_bytes_class(label, epoch_seed, class);
let key = crate::seed::seed_words_from_bytes(day_bytes);
let mut dataset = DatasetSource::from_key(key, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2);
let shape = Shape::for_class_day(&class, days_since_genesis);
let dataset_log2 = if class.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 };
let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape);
dataset.key_bytes = day_bytes.to_vec();
dataset.attach_hot_for(&program);
Self { program, dataset }
}
/// The chain's shape with the program class (Counter ASIC 2.0, 5 October 2026): what the node's engine and the
/// miner's pack export build from the seeds a block template carries. Class v2 is [`Epoch::from_seed_bytes`]
/// exactly (the era bytes are ignored and not recorded); class v3 draws from [`crate::generator::V3_CLASS`]
/// with generator version 3 and records the era seed bytes (`E_n`) in the program for the pack.
pub fn from_chain_seeds(epoch_seed: &[u8], day_bytes: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Self {
Self { program: Self::chain_program(epoch_seed, era_bytes, class, label), dataset: Self::chain_dataset(day_bytes, class) }
}
/// The program alone of [`Epoch::from_chain_seeds`] (no cache fill): for an engine that shares the day's cache.
pub fn chain_program(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Program {
crate::generator::generate_from_seed_bytes_program_class(label, epoch_seed, class, era_bytes)
}
/// The day's cache and dataset of [`Epoch::from_chain_seeds`], the one entry the node's engine builds a day
/// cache through. The class is an argument because the Counter ASIC 2.0 integration gives class v3 its own item
/// construction (the mixer multiplier) and cache size schedule (ca2-mixer); today both classes build the day of
/// [`Epoch::from_seed_bytes`], and the engine keys its day caches on `(day, class)` so the two never share one.
pub fn chain_dataset(day_bytes: &[u8], class: ProgramClass) -> DatasetSource {
Self::chain_dataset_day(day_bytes, class, 0, DEFAULT_DATASET_LOG2)
}
/// [`Epoch::chain_dataset`] with the day's position since genesis and the network's genesis dataset size: the
/// entry the node's engine and the miner's export build every day cache through, so the cache growth schedule
/// of spec 01 section 1.13.3 has one place to act (ca2-mixer, 5 October 2026, `docs/plans/mixer-x4.md`): the
/// class's load class gives the mixer multiplier and whether the growth rule applies (`Shape::for_class_day`);
/// under the rule the cache is `2^memhard::cache_log2_words(d)` words and the dataset
/// `2^memhard::dataset_log2_words(genesis_dataset_log2, d)`; without it (class v2) the cache is 2^26 words and
/// the dataset the genesis size on every day. `days_since_genesis` is [`days_since_genesis`] of the block's and
/// the genesis header's day indices.
pub fn chain_dataset_day(day_bytes: &[u8], class: ProgramClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> DatasetSource {
let lc = class.load_class();
let shape = Shape::for_class_day(&lc, days_since_genesis);
let dataset_log2 = if lc.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 };
let key = crate::seed::seed_words_from_bytes(day_bytes);
let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape);
dataset.key_bytes = day_bytes.to_vec();
dataset
}
/// The 32 hashes of the warp starting at `base_nonce`.
pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] {
hash_warp(&self.program, base_nonce, &self.dataset)
@ -356,6 +685,190 @@ mod tests {
assert_eq!(ds.word(0x0fffffff), 0xf78c84a4);
}
/// Read-width experiment: the fold for one word is a plain xor; a wide fetch hands each lane the words the
/// scalar path would; two distinct lines give two distinct maps of dst (one point suffices as a smoke check).
#[test]
fn fold_and_wide_fetch() {
assert_eq!(fold_words(0x1234_5678, &[0xdead_beef]), 0x1234_5678 ^ 0xdead_beef);
let w = [1u32, 2, 3, 4];
let x = fold_words(7, &w);
let mut y: u32 = 7 ^ 1;
for &v in &w[1..] {
y = y.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ v;
}
assert_eq!(x, y);
assert_ne!(fold_words(7, &[1, 2, 3, 4]), fold_words(7, &[1, 2, 3, 5]));
let ds = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20);
let mut base = [0u32; LANES];
for (k, b) in base.iter_mut().enumerate() {
*b = ((k as u32).wrapping_mul(0x9E37_79B1) & ds.mask) & !15;
}
let mut out = [[0u32; 16]; LANES];
let items = ds.fetch_wide(&base, 16, &mut out, Layout::LINEAR);
assert!(items >= 1 && items <= LANES);
for k in 0..LANES {
for j in 0..16 {
assert_eq!(out[k][j], ds.word(base[k] + j as u32), "lane {k} word {j}");
}
}
let mut base4 = base;
for b in base4.iter_mut() {
*b += 8;
}
let items4 = ds.fetch_wide(&base4, 4, &mut out, Layout::LINEAR);
assert_eq!(items4, items);
for k in 0..LANES {
for j in 0..4 {
assert_eq!(out[k][j], ds.word(base4[k] + j as u32));
}
}
}
/// A wide-load program interprets identically on the closed form and through the memory-hard path's fold
/// (the same fold code), and a mixed-class epoch builds and hashes.
/// Variant 5: a fill word is deterministic, a rewrite changes the slot, and a second read of a written slot
/// returns the rewrite, not the fill.
#[test]
fn scratch_model() {
let seed = [1u32, 2, 3, 4, 5, 6, 7, 8];
assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1));
let mut m = ScratchModel::new(256);
let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)];
let x = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x, fold_words(0xabcd, &w));
let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w)));
assert_eq!(m.reads, 2);
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128));
assert_eq!(e.program.scratch_ops_per_hash(), 32);
assert_eq!(e.hash_warp(0), e.hash_warp(0));
}
/// Era layout: the load address stays inside the site's window and below the mask at every dataset size (the
/// window floor of 2^26 words clamps the shrink), the interleaved memory-hard dataset reads item(t(w))[j(w)] and
/// is the same prefix at 2^20 and 2^22 words, the wide fetch agrees word for word, and an era epoch hashes
/// deterministically through the interpreter and the single-nonce API.
#[test]
fn era_windows_layout_and_epochs() {
let eb = EraParams::test_era_bytes("igneum-era-test/1");
let c = LoadClass::era(LoadClass::V2, &eb, &[1]);
let e = c.era.unwrap();
let mut ins = Instr { op: Op::Load, dst: 0, src: 1, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 2, off: 3 };
let mut s = crate::seed::SplitMix64::new(7);
for log2 in [20u32, 26, 27, 28, 29] {
let mask = (1u64 << log2) as u32 - 1;
let k = ins.win.min(log2.saturating_sub(26) as u8) as u32;
for _ in 0..1000 {
let x = s.next() as u32;
let idx = load_index(Some(&e), &ins, x, mask, log2);
assert!(idx <= mask);
let (wm, off) = window(&ins, mask, log2);
assert_eq!(idx & !wm, off, "log2 {log2}");
assert_eq!(wm, mask >> k);
assert_eq!(idx, ((x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot) & wm) | off) & mask);
}
}
ins.win = 0;
assert_eq!(load_index(None, &ins, 0xdead_beef, 0x0fff_ffff, 28), 0xdead_beef & 0x0fff_ffff);
// the interleaved dataset: one day cache, the layout per program
let l = e.layout();
assert_eq!(l.pos, [1, 3, 8, 13]);
let small = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20);
let big = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 22);
let m = small.memhard().unwrap();
for w in [0u32, 1, 4, 5, 255, 256, 4095, 8192, 0x0f_ffff] {
let (t, j) = l.split(w);
assert_eq!(small.word_at(l, w), crate::memhard::derive_item(t, &m.params, &m.cache)[j as usize], "w {w}");
assert_eq!(small.word_at(l, w), big.word_at(l, w), "prefix at w {w}");
assert_eq!(small.word(w), small.word_at(Layout::LINEAR, w));
}
let mut idx = [0u32; LANES];
for (k, i) in idx.iter_mut().enumerate() {
*i = (k as u32).wrapping_mul(0x9E37_79B1) & small.mask;
}
let mut out = [0u32; LANES];
small.fetch(&idx, &mut out, l);
for k in 0..LANES {
assert_eq!(out[k], small.word_at(l, idx[k]));
}
// a 16-byte era: the wide fetch keeps a lane's four words in one item
let eb3 = EraParams::test_era_bytes("igneum-era-test/3");
let c3 = LoadClass::era(LoadClass::fixed(4, 16), &eb3, &[4]);
let l3 = c3.layout();
assert_eq!(l3.pos[..2], [0, 1]);
let mut base = [0u32; LANES];
for (k, b) in base.iter_mut().enumerate() {
*b = ((k as u32).wrapping_mul(0x9E37_79B1) & small.mask) & !3;
}
let mut wide = [[0u32; 16]; LANES];
small.fetch_wide(&base, 4, &mut wide, l3);
for k in 0..LANES {
for j in 0..4 {
assert_eq!(wide[k][j], small.word_at(l3, base[k] + j as u32), "lane {k} word {j}");
}
}
// era epochs hash deterministically, differ per era, and the single-nonce API agrees with the warp
let mut seen = std::collections::HashSet::new();
for (n, c) in [(1u64, c), (3, c3)] {
let ep = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::MemoryHard, 20, c);
assert_eq!(ep.dataset_word(5), ep.dataset.word_at(c.layout(), 5));
let a = ep.hash_warp(64);
assert_eq!(a, ep.hash_warp(64));
assert_eq!(ep.hash(64 + 5), a[5]);
assert!(seen.insert(a[0]), "era {n}");
let closed = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c);
assert_ne!(closed.hash_warp(64), a);
}
}
/// Hot-table experiment: an epoch of a hot class carries the table, hashes deterministically and differs from
/// version 2; the reference interpreter agrees with a hand-stepped hot load; a hot program without its table is
/// refused.
#[test]
fn hot_epochs_hash() {
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(32, 4));
let h = e.dataset.hot.as_ref().expect("the epoch fills the hot table");
assert_eq!(h.n_words(), 1 << 23);
assert_eq!(h.key, crate::memhard::hot_key(b"igneum-genesis"));
let v2 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::V2);
let a = e.hash_warp(0);
assert_eq!(a, e.hash_warp(0));
assert_ne!(a, v2.hash_warp(0));
assert_ne!(a[0], a[1]);
// the same program under a 64 MiB table reads other words
let e64 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(64, 4));
assert_eq!(e64.program.instrs, e.program.instrs);
assert_ne!(e64.hash_warp(0), a);
// from seed bytes, the chain's shape, with a hot class
let genesis = crate::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap();
let ec = Epoch::from_seed_bytes_class(&genesis, &crate::bind::day_bytes(20_730), "devnet", LoadClass::hot(32, 2));
assert_eq!(ec.dataset.hot.as_ref().unwrap().key, crate::memhard::hot_key(&genesis));
assert_eq!(ec.hash_warp(0), ec.hash_warp(0));
}
#[test]
#[should_panic(expected = "needs the epoch's hot table")]
fn hot_program_without_a_table_is_refused() {
let p = generate_class("igneum-genesis", LoadClass::hot(32, 4));
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 20);
let _ = hash_warp(&p, 0, &ds);
}
#[test]
fn wide_class_epochs_hash() {
for name in ["w16", "w64x4", "50,35,15"] {
let c = LoadClass::parse(name).unwrap();
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c);
assert_eq!(e.program.class, c);
let a = e.hash_warp(0);
let b = e.hash_warp(0);
assert_eq!(a, b);
assert_ne!(a[0], a[1]);
}
}
#[test]
fn closed_form_genesis_vector_lane0() {
// Generator v2 vectors (4 October 2026), proto-cuda/packs/igneum-genesis/vectors.json.

247
igneum-pow/tests/mixer.rs Normal file
View file

@ -0,0 +1,247 @@
//! The class v3 dataset construction (mixer x4, cache growth option C; `docs/plans/mixer-x4.md`): the soundness
//! runs the brief asks for, on the CPU, with the packs for the GPU runs written on request.
//!
//! 1. Fuzz: `IGNEUM_MIXER_FUZZ` (default 200) programs through the seam (`ProgramClass::V3`), the contract on every
//! instruction (the v2 program of the seed, instruction for instruction), 4 units each across the 32-bit range
//! including the wrap, interpreted twice on the CPU; with `IGNEUM_MIXER_PACKS_OUT=<dir>` every program is written
//! as a pack with its 4 bases in vectors.json for `packbench` and the OpenCL host (the Metal fuzz).
//! 2. Stats: bit balance and single-bit-flip avalanche of the v3 hash against v2 on the same programs and nonces.
//! 3. Edge: the dataset at word 0, word MASK and the item boundary, derived through the interpreter's fetch path and
//! by hand at every multiplier 1, 2, 4, 8, on a small cache.
//! 4. Determinism: two independent epochs of the same seed and day agree on every vector and every emitted file.
use igneum_pow::emit::{export_pack, vectors_json};
use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION_V3, INSTR_COUNT, V3_CLASS};
use igneum_pow::memhard::{derive_item, mixer, round_key, Cache, MixParams, Shape};
use igneum_pow::seed::{day_key, SplitMix64};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
use std::collections::HashMap;
use std::path::PathBuf;
const DAY: &str = "2026-10-03";
fn contract(p: &Program, seed: &str, class: LoadClass) {
if class == V3_CLASS {
assert_eq!(p.generator, GENERATOR_VERSION_V3);
}
assert_eq!(p.class, class);
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16);
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a v3 load reads one word");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
let v2 = generate_from_seed_bytes(seed, seed.as_bytes());
assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed under the v3 construction");
assert_eq!(p.attempt, v2.attempt);
}
/// Write a pack whose vectors.json carries `bases` instead of the three standard bases (packbench and the OpenCL
/// host check every unit standalone and the ones inside the batch window).
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
}
#[test]
fn fuzz_v3_programs_cpu() {
let n: usize = std::env::var("IGNEUM_MIXER_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
let out = std::env::var("IGNEUM_MIXER_PACKS_OUT").ok().map(PathBuf::from);
// IGNEUM_MIXER_CLASS=mx8 fuzzes the x8 candidate as a load class (generator 2 with the class in the id); the
// default is V3_CLASS through the seam
let class = std::env::var("IGNEUM_MIXER_CLASS").ok().map(|s| LoadClass::parse(&s).expect("a load class")).unwrap_or(V3_CLASS);
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d6d); // "igneum-m"
let shape = Shape::for_class(&class);
assert_eq!(shape.cache_log2_words, 26);
assert!(shape.mixer_mult > 1);
// one memory-hard source per dataset size (the 256 MiB cache fill is 0.2 s each)
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tlog2\tprogram_id\tbases\n");
let mut units = 0usize;
let mut wraps = 0usize;
for i in 0..n {
let seed = format!("igneum-mixer-fuzz/{i}");
let p = if class == V3_CLASS {
generate_from_seed_bytes_program_class(&seed, seed.as_bytes(), ProgramClass::V3, None)
} else {
generate_from_seed_bytes_class(&seed, seed.as_bytes(), class)
};
contract(&p, &seed, class);
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, log2, shape));
let e = Epoch { program: p, dataset: ds };
for &b in &bases {
let r1 = e.interpret_warp(b);
let r2 = e.interpret_warp(b);
assert_eq!(r1.hashes, r2.hashes);
assert!(r1.items_derived >= 120 * 32 / 32 && r1.items_derived <= 4_096, "{seed}: {} items", r1.items_derived);
units += 1;
}
if let Some(dir) = &out {
let pack_name = format!("fuzz-{i:03}-{}-l{log2}", class.name());
write_pack_with_bases(&dir.join(&pack_name), &e, DAY, &bases, "igneum-pow tests/mixer.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{log2}\t{:016x}\t{}\n",
e.program.program_id(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
}
mh.insert(log2, e.dataset);
}
println!("fuzz: {n} {} programs, {units} units on the CPU, {wraps} units in the top 256 nonces", class.name());
assert_eq!(units, 4 * n);
assert_eq!(wraps, n);
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// Bit balance and avalanche of the v3 hash beside v2 on the same program (the TESTS.md section 3 shape, on the
/// CPU, 2^13 nonces per seed): every output bit within 5 sigma of half ones; a single nonce-bit flip moves 50 percent
/// of the output bits within 2 points; no duplicate among the outputs.
#[test]
fn stats_v3_against_v2() {
let n_warps = 256usize; // 8,192 nonces
for seed in ["igneum-genesis", "igneum-genesis/stats1"] {
let v3 = Epoch {
program: generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, None),
dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 24, Shape::for_class(&V3_CLASS)),
};
let v2 = Epoch::new(seed, DAY, DatasetMode::MemoryHard, 24);
for (name, e) in [("v3", &v3), ("v2", &v2)] {
let mut ones = [0u64; 64];
let mut outs = Vec::with_capacity(n_warps * 32);
for w in 0..n_warps {
let h = e.hash_warp(w as u32 * 32);
for &x in &h {
outs.push(x);
for b in 0..64 {
ones[b] += (x >> b) & 1;
}
}
}
let total = (n_warps * 32) as f64;
let sigma = (total / 4.0).sqrt();
for (b, &c) in ones.iter().enumerate() {
let z = (c as f64 - total / 2.0).abs() / sigma;
assert!(z < 5.0, "{seed} {name}: bit {b} ones {c} of {total}, z {z:.2}");
}
// avalanche: flip one bit of the nonce within the unit (lanes 0..31 differ in the low 5 bits) and across
// units (bit 5 and up): compare lane l of unit u with lane l ^ (1 << k) and with unit u ^ (1 << k)
let mut flips = 0u64;
let mut moved = 0u64;
for w in 0..64usize {
let h = e.hash_warp(w as u32 * 32);
for k in 0..5 {
for l in 0..32usize {
moved += (h[l] ^ h[l ^ (1 << k)]).count_ones() as u64;
flips += 1;
}
}
let h2 = e.hash_warp((w ^ 1) as u32 * 32);
for l in 0..32usize {
moved += (h[l] ^ h2[l]).count_ones() as u64;
flips += 1;
}
}
let avg = moved as f64 / flips as f64 / 64.0 * 100.0;
assert!((avg - 50.0).abs() < 2.0, "{seed} {name}: avalanche {avg:.2} percent");
outs.sort_unstable();
let dups = outs.windows(2).filter(|p| p[0] == p[1]).count();
assert_eq!(dups, 0, "{seed} {name}: duplicate outputs");
println!("{seed} {name}: {} outputs, avalanche {avg:.2} percent, worst bit z {:.2}", outs.len(), ones.iter().map(|&c| (c as f64 - total / 2.0).abs() / sigma).fold(0.0, f64::max));
}
assert_ne!(v3.hash_warp(0), v2.hash_warp(0));
}
}
/// The dataset edges under every multiplier on a small cache: word 0, word MASK, the last word of item 0 and the
/// first of item 1, through `DatasetSource::word` and by hand.
#[test]
fn edge_items_every_multiplier() {
let key = day_key(DAY);
let cache = Cache::fill_log2(key, 14);
for m in [1u32, 2, 4, 8] {
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14 });
let by_hand = |t: u32| -> [u32; 16] {
let mut s = [0u32; 16];
s[..8].copy_from_slice(&key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
for r in 0..8usize {
for j in 0..m as usize {
mixer(&mut s, round_key(r * m as usize + j), &mp);
}
let line = cache.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
for j in 0..m as usize {
mixer(&mut s, round_key(8 * m as usize + j), &mp);
}
s
};
for t in [0u32, 1, 0x0fff_ffff, 0xffff_ffff] {
assert_eq!(derive_item(t, &mp, &cache), by_hand(t), "m {m} item {t}");
}
}
// the interpreter's fetch path at the genesis cache: words 0, 15, 16 and MASK of a 2^20-word dataset agree with
// the item derivation, under v3
let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape::for_class(&V3_CLASS));
let m = ds.memhard().unwrap();
for w in [0u32, 15, 16, 17, ds.mask - 1, ds.mask] {
assert_eq!(ds.word(w), derive_item(w >> 4, &m.params, &m.cache)[(w & 15) as usize]);
assert_eq!(ds.word(w), m.word(w));
}
// a load at an out-of-range register masks to the dataset: the word at mask + 1 is the word at 0
assert_eq!(ds.word(ds.mask.wrapping_add(1)), ds.word(0));
}
/// Two independent epochs of the same seed and day: every vector and every emitted file identical; the pinned v3
/// pack is what a third export writes.
#[test]
fn determinism_v3() {
let build = || Epoch {
program: generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, None),
dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 28, Shape::for_class(&V3_CLASS)),
};
let a = build();
let b = build();
let pa = export_pack(&a, DAY, "a");
let pb = export_pack(&b, DAY, "a");
assert_eq!(pa.outs, pb.outs);
assert_eq!(pa.vectors, pb.vectors);
assert_eq!(pa.files, pb.files);
for (w, warp) in [(0u32, 0usize), (4096, 1), (1_000_000, 2)] {
assert_eq!(a.hash_warp(w), pa.outs[warp]);
}
let dir = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer/mx8-genesis");
for (name, text) in &pa.files {
if name == "vectors.json" || name == "vectors.h" {
continue; // the source string differs ("a" here)
}
let on_disk = std::fs::read_to_string(dir.join(name)).unwrap();
assert_eq!(&on_disk, text, "{name}");
}
}

View file

@ -5,28 +5,52 @@
//!
//! Packs: igneum-genesis-mh and igneum-devnet-v4-epoch0 (memory-hard; the latter from the devnet genesis hash as
//! the epoch seed and the day bytes of 2026-10-04), igneum-genesis and igneum-hourly (closed-form dataset,
//! interpreter regression only).
//! interpreter regression only); and, under `proto-cuda/packs-ca2-mixer/`, the class v3 packs mx8-genesis and
//! mx8-devnet-epoch0 (Counter ASIC 2.0, 5 October 2026: generator 3 on `V3_CLASS` = mixer x8 with the cache growth
//! rule, decided 22:05 UTC under the delegated rule; the same seeds and days as the two memory-hard v2 packs, so the
//! v2 program and cache carry over and only the dataset words and the hashes change) and the x4 candidate's packs
//! mx4-genesis and mx4-devnet-epoch0 (generator 2 with the load class in the id, the record of the x4 rows).
use igneum_pow::accept;
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_program,
cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_memhard_for, metal_program,
metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource,
};
use igneum_pow::generator::{generate_from_seed_bytes, Op, GENERATOR_VERSION, LOAD_SLOTS};
use igneum_pow::memhard::CACHE_WORDS;
use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS};
use igneum_pow::memhard::{Shape, CACHE_WORDS};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
use serde_json::Value;
use std::path::PathBuf;
use std::sync::OnceLock;
const PACKS: [&str; 4] = ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "igneum-genesis", "igneum-hourly"];
const PACKS: [&str; 8] = [
"igneum-genesis-mh",
"igneum-devnet-v4-epoch0",
"igneum-genesis",
"igneum-hourly",
"mx8-genesis",
"mx8-devnet-epoch0",
"mx4-genesis",
"mx4-devnet-epoch0",
];
/// The class v3 packs (generator 3 through the seam).
const PACKS_V3: [&str; 2] = ["mx8-genesis", "mx8-devnet-epoch0"];
fn packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs")
}
/// The directory a pack lives in: the class v3 packs under packs-ca2-mixer, the rest under packs.
fn pack_dir(pack: &str) -> PathBuf {
if pack.starts_with("mx4-") || pack.starts_with("mx8-") {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer").join(pack)
} else {
packs_dir().join(pack)
}
}
fn read(pack: &str, file: &str) -> String {
let p = packs_dir().join(pack).join(file);
let p = pack_dir(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
@ -62,9 +86,22 @@ fn epoch(pack: &str) -> &'static Epoch {
_ => DatasetMode::ClosedForm,
};
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let program = generate_from_seed_bytes(seed, &seed_bytes);
// a class v3 pack: generator 3 on V3_CLASS through the seam, the era bytes it records, the dataset
// in the class's shape on day 0 (the growth rule's genesis cache: every pinned pack is a day-0 size)
let program = match j["generator"].as_u64().unwrap() as u32 {
GENERATOR_VERSION_V3 => {
let era = j.get("era_seed_bytes").map(unhex);
generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref())
}
// a generator 2 pack with a load class (the x4 candidate's packs): the class from program.json
_ => match j.get("load_class").and_then(|c| c.as_str()) {
Some(c) => generate_from_seed_bytes_class(seed, &seed_bytes, LoadClass::parse(c).expect("a load class name")),
None => generate_from_seed_bytes(seed, &seed_bytes),
},
};
let shape = Shape::for_class(&program.class);
let mut dataset =
DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2);
DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2, shape);
dataset.key_bytes = day_bytes;
(p.to_string(), Epoch { program, dataset })
})
@ -81,7 +118,8 @@ fn check_program_json(pack: &str) {
let j = json(pack, "program.json");
let p = &epoch(pack).program;
assert_eq!(j["format"].as_str().unwrap(), "igneum-program-pack-3");
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION, "{pack}: generator version");
assert_eq!(j["generator"].as_u64().unwrap() as u32, p.generator, "{pack}: generator version");
assert_eq!(p.generator, if PACKS_V3.contains(&pack) { GENERATOR_VERSION_V3 } else { GENERATOR_VERSION });
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
let sw: Vec<u32> = j["seed_words"].as_array().unwrap().iter().map(hex32).collect();
@ -129,7 +167,7 @@ fn genesis_program_shape() {
#[test]
fn mixer_params_match_pack() {
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0"] {
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "mx8-genesis", "mx8-devnet-epoch0", "mx4-genesis", "mx4-devnet-epoch0"] {
let j = json(pack, "program.json");
let mp = &epoch(pack).dataset.memhard().unwrap().params;
let key: Vec<u32> = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect();
@ -166,19 +204,21 @@ fn cache_matches_vectors() {
fn check_dataset_words(pack: &str) {
let v = json(pack, "vectors.json");
let ds = &epoch(pack).dataset;
let e = epoch(pack);
let ds = &e.dataset;
// a pack's self-test words are read under its program's layout (the era layout; linear for every v2 pack)
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(ds.word(i as u32), *h, "{pack}: dataset[{i}]");
assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]");
}
let last_index = v["dataset_last_index"].as_u64().unwrap() as u32;
assert_eq!(last_index, ds.mask);
assert_eq!(ds.word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
assert_eq!(e.dataset_word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
let samples = v["dataset_samples"].as_array().unwrap();
assert_eq!(samples.len(), 64);
for s in samples {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(ds.word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
}
}
@ -261,12 +301,15 @@ fn check_sources(pack: &str) {
assert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset));
if let Some(mp) = mp {
assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
assert_same_text(pack, "memhard.metal", &metal_memhard(mp));
assert_same_text(pack, "memhard.metal", &igneum_pow::emit::metal_memhard_layout(mp, p.class.layout()));
}
let got = program_json(p, &day, &e.dataset);
assert_same_text(pack, "program.json", &got);
let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON");
// Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them.
// Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them (a class with
// the era layout inside has the era form instead: `((rotl_imm(rN * M, R) & WM) | OFF) & mask`, checked by
// era_emitted_sources_match_and_loads_have_the_era_form over the era packs, and here by the same count).
let era_load = if p.class.era.is_some() { "((rotl_imm(r" } else { "" };
for (file, load, masked) in [
("kernel.cu", "ds[r", " & mask]"),
("kernel_bound.cu", "ds[r", " & mask]"),
@ -274,6 +317,7 @@ fn check_sources(pack: &str) {
("program_bound.metal", "dataset[r", " & MASK]"),
] {
let text = read(pack, file);
let load = if era_load.is_empty() { load } else { era_load };
assert_eq!(text.matches(load).count(), LOAD_SLOTS, "{pack}/{file}: 16 loads");
assert_eq!(text.matches(masked).count(), LOAD_SLOTS, "{pack}/{file}: 16 masked loads");
}
@ -287,6 +331,85 @@ fn emitted_sources_match_all_packs() {
}
/// The whole pack as `export` writes it: vectors.json and vectors.h match, and the file list is the full set.
/// The class v3 packs (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): generator 3 on V3_CLASS = mx8; the program of
/// each is the v2 program of the same seed instruction for instruction (v2 loads take no width roll); the cache is
/// the v2 cache (day 0 of the growth rule: 2^26 words, the same FNV-1a 64); the dataset words differ from v2's;
/// program.json, program.h and the emitted memhard core say so; the id carries generator 3.
#[test]
fn v3_packs_are_the_v2_seeds_under_mixer_x8() {
assert_eq!(V3_CLASS.name(), "mx8");
assert_eq!(V3_CLASS.mixer_mult, 8);
assert!(V3_CLASS.growth);
assert_eq!(LoadClass { mixer_mult: 4, ..V3_CLASS }, LoadClass::MX4, "the x4 candidate differs from v3 in the multiplier alone");
for (v3, v2) in [("mx8-genesis", "igneum-genesis-mh"), ("mx8-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
let e3 = epoch(v3);
let e2 = epoch(v2);
let j = json(v3, "program.json");
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
assert!(j["load_class"].as_str().unwrap().starts_with("mx8"), "{v3}: mx8, or mx8 with the era inside");
assert_eq!(j["mixer_mult"].as_u64().unwrap(), 8);
assert_eq!(j["cache_growth"].as_bool().unwrap(), true);
assert_eq!(j["dataset"]["mixer_mult"].as_u64().unwrap(), 8);
assert_eq!(j["dataset"]["cache"]["log2_words"].as_u64().unwrap(), 26);
assert_eq!(e3.program.generator, GENERATOR_VERSION_V3);
if e3.program.era_bytes.is_some() {
// the era layout composed into class v3 (docs/plans/era-layout.md, 5 October 2026): a chain pack carries an
// era, so its class is V3_CLASS with the era drawn inside and its stream takes two window draws per
// instruction; the v2 program carries over only in the seed, the attempt and the day
assert_eq!(LoadClass { era: None, ..e3.program.class }, V3_CLASS, "{v3}: the composed class");
assert!(e3.program.class.era.is_some());
assert_ne!(e3.program.instrs, e2.program.instrs, "{v3}: the era windows change the stream");
} else {
assert_eq!(e3.program.class, V3_CLASS);
assert_eq!(e3.program.instrs, e2.program.instrs, "{v3}: the v2 program under the v3 construction");
}
assert_eq!(e3.program.seed, e2.program.seed);
assert_eq!(e3.program.attempt, e2.program.attempt);
assert_ne!(e3.program.program_id(), e2.program.program_id());
assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt));
let m3 = e3.dataset.memhard().unwrap();
let m2 = e2.dataset.memhard().unwrap();
assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26 });
assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0");
assert_eq!(m3.params.rot, m2.params.rot);
assert_eq!(e3.dataset.log2_words, 28);
assert_ne!(e3.dataset.word(0), e2.dataset.word(0), "{v3}: the dataset words differ");
assert_ne!(e3.hash_warp(0), e2.hash_warp(0));
let h = read(v3, "program.h");
assert!(h.contains("#define IGNEUM_GENERATOR 3\n"));
assert!(h.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
assert!(h.contains("#define IGNEUM_MIXER_MULT 8"));
assert!(h.contains("#define IGNEUM_CACHE_GROWTH 1"));
assert!(h.contains("#define IGNEUM_CACHE_LOG2_WORDS 26\n"));
assert!(h.contains("#define IGNEUM_LOAD_CLASS \"mx8"), "mx8, or mx8 with the era inside");
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
let text = read(v3, file);
assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u))").count(), 1, "{v3}/{file}");
assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u))").count(), 1, "{v3}/{file}");
}
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
let text = read(v2, file);
assert_eq!(text.matches("j < 8u").count(), 0, "{v2}/{file}: the v2 text has no multiplier loop");
}
}
// the devnet v3 pack records era 0's stand-in, the devnet genesis hash
let j = json("mx8-devnet-epoch0", "program.json");
assert_eq!(unhex(&j["era_seed_bytes"]), unhex(&j["seed_bytes"]));
assert!(read("mx8-devnet-epoch0", "program.h").contains("#define IGNEUM_ERA_SEED_HEX \"edc4fa844da9dc98"));
assert!(json("mx8-genesis", "program.json").get("era_seed_bytes").is_none());
// the x4 candidate's packs: generator 2, the class in the id, the v2 program of the seed, mixer x4
for (x4, v2) in [("mx4-genesis", "igneum-genesis-mh"), ("mx4-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
let e4 = epoch(x4);
let j = json(x4, "program.json");
assert_eq!(e4.program.generator, GENERATOR_VERSION);
assert_eq!(j["load_class"].as_str().unwrap(), "mx4");
assert_eq!(e4.program.class, LoadClass::MX4);
assert_eq!(e4.program.instrs, epoch(v2).program.instrs);
assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26 });
assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))"));
}
}
fn check_export(pack: &str) {
let e = epoch(pack);
let v = json(pack, "vectors.json");
@ -311,7 +434,7 @@ fn check_export(pack: &str) {
expected.extend(["memhard.h", "memhard.metal"]);
}
assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::<Vec<_>>(), expected);
let mut on_disk: Vec<String> = std::fs::read_dir(packs_dir().join(pack))
let mut on_disk: Vec<String> = std::fs::read_dir(pack_dir(pack))
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| !n.starts_with('.'))
@ -350,3 +473,429 @@ fn devnet_pack_is_the_chain_derivation() {
assert_eq!(e.program.instrs, epoch("igneum-devnet-v4-epoch0").program.instrs);
assert_eq!(e.hash_warp(0), epoch("igneum-devnet-v4-epoch0").hash_warp(0));
}
// ---------------------------------------------------------------------------------------------------------
// Era layout packs (5 October 2026, docs/plans/era-layout.md): proto-cuda/packs-ca2-era/era-<n>, n in 0..5, the
// devnet epoch seed and day bytes under the era class of test era seed igneum-era-test/<n>, the width pinned at
// 4 bytes (allowed_widths in program.json). Checked like the pinned packs, plus the one load form of 1.3 by text search.
// ---------------------------------------------------------------------------------------------------------
use igneum_pow::generator::{generate_era, EraParams, V3_ALLOWED};
const ERA_PACKS: [&str; 6] = ["era-0", "era-1", "era-2", "era-3", "era-4", "era-5"];
fn era_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-era")
}
fn era_read(pack: &str, file: &str) -> String {
let p = era_packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn era_json(pack: &str, file: &str) -> Value {
serde_json::from_str(&era_read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
/// The era class of a pack: the era bytes from program.json (`era_seed_bytes`, the chain's `E_n`; the test seed
/// `igneum-era-test/<n>` of the pack's number gives the same bytes), the allowed set from `era.allowed_widths`
/// (class v3's `V3_ALLOWED`); the pack's recorded stream words and draw must be the class's.
fn era_class(pack: &str) -> LoadClass {
let j = era_json(pack, "program.json");
let n: u64 = pack.trim_start_matches("era-").parse().unwrap();
let eb = unhex(&j["era_seed_bytes"]);
assert_eq!(eb, EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec(), "{pack}: the era bytes of test seed {n}");
let allowed: Vec<u8> = j["era"]["allowed_widths"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect();
assert_eq!(allowed, V3_ALLOWED.to_vec(), "{pack}: class v3's width set");
// the measurement packs of 5 October 2026: the era layout over version 2's construction (mixer x1, the genesis
// cache), generator 3 and the era bytes recorded; the chain's class v3 composes the same draw over LoadClass::MX4
// (generator tests era_programs_are_accepted and program_classes), and the integration re-exports these packs
let c = LoadClass::era(LoadClass::V2, &eb, &allowed);
let e = c.era.unwrap();
let words: Vec<u32> = j["era"]["seed_words"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(e.words.to_vec(), words, "{pack}: era seed words");
assert_eq!(e.width_words as u64, j["era"]["width_words"].as_u64().unwrap(), "{pack}: width");
assert_eq!(e.stride_mul, hex32(&j["era"]["stride_mul"]), "{pack}: stride mul");
assert_eq!(e.stride_rot as u64, j["era"]["stride_rot"].as_u64().unwrap(), "{pack}: stride rot");
let pos: Vec<u8> = j["era"]["interleave"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect();
assert_eq!(e.pos.to_vec(), pos, "{pack}: interleave");
c
}
fn era_epoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
ERA_PACKS
.iter()
.map(|p| {
let j = era_json(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let class = era_class(p);
assert_eq!(log2, igneum_pow::verify::DEFAULT_DATASET_LOG2);
let eb = unhex(&j["era_seed_bytes"]);
let program = generate_era(seed, &seed_bytes, LoadClass::V2, &eb, &V3_ALLOWED);
assert_eq!(program.class, class, "{p}: the pack's era class");
assert_eq!(program.generator, GENERATOR_VERSION_V3);
assert_eq!(program.era_bytes.as_deref(), Some(&eb[..]));
let mut dataset = DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
/// program.json of an era pack: generator, attempt, id, class, every instruction with width, win and off, and the
/// program passes the acceptance rule.
#[test]
fn era_program_json_matches() {
for pack in ERA_PACKS {
let j = era_json(pack, "program.json");
let p = &era_epoch(pack).program;
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION_V3, "{pack}: a class v3 pack");
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
assert_eq!(j["load_class"].as_str().unwrap(), p.class.name(), "{pack}: class");
assert_eq!(j["bytes_per_hash"].as_u64().unwrap() as usize, p.bytes_per_hash());
assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS);
assert!(accept::check(p).is_ok(), "{pack}: acceptance");
let instrs = j["instructions"].as_array().unwrap();
assert_eq!(instrs.len(), p.instrs.len());
for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() {
assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op");
assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64);
assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64);
assert_eq!(hex32(&ji["imm"]), ins.imm);
assert_eq!(ji["width"].as_u64().unwrap(), ins.width as u64, "{pack} #{k} width");
assert_eq!(ji["win"].as_u64().unwrap(), ins.win as u64, "{pack} #{k} win");
assert_eq!(ji["off"].as_u64().unwrap(), ins.off as u64, "{pack} #{k} off");
if ins.op == Op::Load {
assert_eq!(ins.width, p.class.era.unwrap().width_words);
assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win));
}
}
}
}
/// The six era packs are the same program seed under six draws: the instruction lists agree, the widths and layouts
/// follow the draw, and the dataset words differ from the linear layout exactly when the interleave is not linear.
#[test]
fn era_packs_share_the_program_and_differ_in_layout() {
let linear = &epoch("igneum-devnet-v4-epoch0").dataset;
for pack in ERA_PACKS {
let e = era_epoch(pack);
assert_eq!(e.program.seed_bytes, epoch("igneum-devnet-v4-epoch0").program.seed_bytes, "{pack}: the devnet seed");
let strip = |p: &igneum_pow::generator::Program| {
p.instrs.iter().map(|i| (i.op, i.dst, i.src, i.src2, i.imm, i.imm2, i.rot, i.bit, i.mask, i.win, i.off)).collect::<Vec<_>>()
};
assert_eq!(strip(&e.program), strip(&era_epoch("era-0").program), "{pack}: same stream as era-0");
let l = e.program.class.layout();
let same_at_1 = (0..64u32).all(|w| e.dataset_word(w * 977 + 1) == linear.word(w * 977 + 1));
assert_eq!(same_at_1, l.is_linear(), "{pack}: layout {:?}", l.pos);
assert_eq!(e.dataset_word(0), linear.word(0), "{pack}: word 0 is item 0 word 0 in every layout");
// the chain's shared day cache serves every era: the day's dataset source is the pinned pack's, bit for bit
assert_eq!(e.dataset.key, linear.key);
assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), linear.memhard().unwrap().cache.fnv1a64());
}
}
#[test]
fn era_dataset_words_and_vectors_match() {
for pack in ERA_PACKS {
let v = era_json(pack, "vectors.json");
let e = era_epoch(pack);
let ds = &e.dataset;
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]");
}
assert_eq!(e.dataset_word(ds.mask), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
for s in v["dataset_samples"].as_array().unwrap() {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
}
assert_eq!(ds.memhard().unwrap().cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
let mut n = 0;
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
n += 1;
}
assert_eq!(e.hash(base + 7), expected[7]);
}
assert_eq!(n, 96, "{pack}");
}
}
/// Every emitted file of every era pack matches the emitters byte for byte, the export reproduces vectors.json and
/// vectors.h, and every dataset load in every hash kernel has the one era form (no plain `ds[rN & mask]` remains).
#[test]
fn era_emitted_sources_match_and_loads_have_the_era_form() {
for pack in ERA_PACKS {
let e = era_epoch(pack);
let day = era_json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string();
let source = era_json(pack, "vectors.json")["source"].as_str().unwrap().to_string();
let out = export_pack(e, &day, &source);
for (name, text) in &out.files {
let want = era_read(pack, name);
assert!(text == &want, "{pack}/{name} differs from the emitter");
}
let mut on_disk: Vec<String> = std::fs::read_dir(era_packs_dir().join(pack))
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| !n.starts_with('.') && n != "seeds.txt")
.collect();
on_disk.sort();
let mut want: Vec<String> = out.files.iter().map(|(n, _)| n.clone()).collect();
want.sort();
assert_eq!(on_disk, want, "{pack}: the pack holds the export's files and seeds.txt only");
let era = e.program.class.era.unwrap();
let mul = format!("0x{:08x}u", era.stride_mul);
for (file, mask) in [
("kernel.cu", "mask"),
("kernel_bound.cu", "mask"),
("kernel.cl", "mask"),
("kernel_bound.cl", "mask"),
("program.metal", "MASK"),
("program_bound.metal", "MASK"),
] {
let text = era_read(pack, file);
let kernels = if file.starts_with("kernel_bound") || file == "kernel.cu" || file == "kernel.cl" || file.starts_with("program") { 1 } else { 1 };
// kernel_bound.cl carries igneum_hash and igneum_hash_bound: two kernels
let kernels = if file == "kernel_bound.cl" { 2 } else { kernels };
let era_form: usize = text
.lines()
.filter(|l| l.contains("rotl_imm(r") && l.contains(&format!(" * {mul}, {}u) & ", era.stride_rot)) && l.contains(&format!(") & {mask}")))
.filter(|l| l.contains("ds[") || l.contains("dataset[") || l.contains("b_ = "))
.count();
assert_eq!(era_form, LOAD_SLOTS * kernels, "{pack}/{file}: {} loads of the era form", LOAD_SLOTS * kernels);
let plain = text.lines().filter(|l| l.contains("ds[r") || l.contains("dataset[r")).count();
assert_eq!(plain, 0, "{pack}/{file}: a load without the era form");
}
// the layout helpers appear exactly when the layout is not linear
let mh = era_read(pack, "memhard.h");
assert_eq!(mh.contains("mh_addr("), !era.layout().is_linear(), "{pack}: memhard.h layout helpers");
}
}
/// An era pack's dataset is a prefix at every size of at least 2^16 words: the 2^20-word source gives the pack's
/// words below 2^20.
#[test]
fn era_dataset_is_a_prefix_at_smaller_sizes() {
for pack in ["era-1", "era-3"] {
let e = era_epoch(pack);
let small = DatasetSource::from_key(e.dataset.key, DatasetMode::MemoryHard, 20);
let l = e.program.class.layout();
for w in [0u32, 1, 2, 3, 16, 255, 4096, 65_535, 65_536, 0x000f_ffff] {
assert_eq!(small.word_at(l, w), e.dataset_word(w), "{pack}: w {w}");
}
}
}
// ---------------------------------------------------------------------------------------------------------
// Hot-table experiment (5 October 2026, docs/plans/hot-table.md): the five packs under proto-cuda/packs-ca2-hot/ are
// pinned the same way (program, vectors, every emitted file byte for byte), plus the hot table's fingerprint and the
// one-form load check: exactly 16 - k masked dataset loads and exactly k hot loads in every hash kernel.
// ---------------------------------------------------------------------------------------------------------
const HOT_PACKS: [&str; 8] = ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "hot32k4a", "hot64k4a", "hot96k4a"];
fn hot_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-hot")
}
fn hread(pack: &str, file: &str) -> String {
let p = hot_packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn hjson(pack: &str, file: &str) -> Value {
serde_json::from_str(&hread(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
/// The epoch of a hot pack from program.json alone: the class from `load_class`, the program from the seed bytes,
/// the dataset from the day bytes, the hot table from the seed bytes (what `Epoch::from_seed_bytes_class` does).
fn hepoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
HOT_PACKS
.iter()
.map(|p| {
let j = hjson(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let class = LoadClass::parse(j["load_class"].as_str().unwrap()).unwrap();
assert_eq!(class.name(), *p, "the pack directory is the class name");
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let program = generate_from_seed_bytes_class(seed, &seed_bytes, class);
let mut dataset =
DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
dataset.attach_hot_for(&program);
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
fn hassert_same_text(pack: &str, file: &str, got: &str) {
let want = hread(pack, file);
if got != want {
let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect());
for i in 0..gl.len().max(wl.len()) {
let g = gl.get(i).copied().unwrap_or("<eof>");
let w = wl.get(i).copied().unwrap_or("<eof>");
if g != w {
panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1);
}
}
panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len());
}
}
#[test]
fn hot_packs_program_and_vectors() {
for pack in HOT_PACKS {
let e = hepoch(pack);
let p = &e.program;
let j = hjson(pack, "program.json");
let h = p.class.hot.unwrap();
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION);
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt);
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
let dataset_slots = if h.added { 16 } else { 16 - h.k as usize };
assert_eq!(j["loads_per_hash"].as_u64().unwrap() as usize, (dataset_slots + h.k as usize) * 8);
assert_eq!(j["hot_table"]["mb"].as_u64().unwrap(), h.mb as u64);
assert_eq!(j["hot_table"]["slots"].as_u64().unwrap(), h.k as u64);
assert_eq!(j["hot_table"]["dataset_slots"].as_u64().unwrap() as usize, dataset_slots);
assert_eq!(j["hot_table"]["words"].as_u64().unwrap() as u32, p.hot_words());
assert_eq!(j["op_mix"]["hot"].as_u64().unwrap(), h.k as u64, "{pack}: k hot instructions");
assert_eq!(j["op_mix"]["load"].as_u64().unwrap() as usize, dataset_slots);
assert_eq!(p.items_per_warp(), dataset_slots * 8 * 32);
assert!(accept::check(p).is_ok(), "{pack}: passes the acceptance rule");
let v2 = &epoch("igneum-genesis-mh").program;
if !h.added {
// replaced form: the version 2 genesis program with k loads redirected (attempt 0 on both)
assert_eq!(p.attempt, v2.attempt);
for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) {
if a.op == Op::Hot {
assert_eq!(b.op, Op::Load);
} else {
assert_eq!(a, b);
}
}
} else {
// added form: 16 + k load slots, so another slot draw and another program; 16 dataset loads stay
assert_ne!(p.instrs, v2.instrs);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16);
assert!(p.instrs.iter().all(|i| i.width == 1));
}
// the hot table: the pack's head, last line and fingerprint
let v = hjson(pack, "vectors.json");
let t = e.dataset.hot.as_ref().unwrap();
assert_eq!(t.n_words(), p.hot_words());
let head: Vec<u32> = v["hot_head"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&t.words()[..16], &head[..]);
let last: Vec<u32> = v["hot_last_line"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&t.words()[t.words().len() - 16..], &last[..]);
assert_eq!(t.fnv1a64(), hex64(&v["hot_fnv1a64"]), "{pack}: hot_fnv1a64");
assert_eq!(t.key, igneum_pow::memhard::hot_key(&p.seed_bytes));
// the cache is the day's, unchanged by the class
assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), 0x48c4f5bf24166b2e);
// 96 vectors
let warps = v["warps"].as_array().unwrap();
assert_eq!(warps.len(), 3);
for w in warps {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
}
assert_eq!(e.hash(base + 5), expected[5]);
}
// the dataset words are the day's
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, hd) in head.iter().enumerate() {
assert_eq!(e.dataset.word(i as u32), *hd);
}
}
// the same k at three sizes: identical programs, three fingerprints, three vector sets
let a = hepoch("hot32k4");
let b = hepoch("hot64k4");
let c = hepoch("hot96k4");
assert_eq!(a.program.instrs, b.program.instrs);
assert_eq!(b.program.instrs, c.program.instrs);
assert_ne!(a.hash_warp(0), b.hash_warp(0));
assert_ne!(b.hash_warp(0), c.hash_warp(0));
}
#[test]
fn hot_packs_emitted_sources_and_load_forms() {
for pack in HOT_PACKS {
let e = hepoch(pack);
let p = &e.program;
let k = p.class.hot.unwrap().k as usize;
let dataset_loads = p.class.dataset_slots();
let day = hjson(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string();
let mp = &e.dataset.memhard().unwrap().params;
hassert_same_text(pack, "kernel.cu", &cuda_kernel(p, Some(mp)));
hassert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, Some(mp)));
hassert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored));
hassert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words));
hassert_same_text(pack, "kernel.cl", &opencl_kernel(p, Some(mp)));
hassert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, Some(mp)));
hassert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset));
hassert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
hassert_same_text(pack, "memhard.metal", &metal_memhard_for(p, mp));
assert_ne!(metal_memhard_for(p, mp), metal_memhard(mp), "{pack}: the hot fill kernel is in memhard.metal");
let got = program_json(p, &day, &e.dataset);
hassert_same_text(pack, "program.json", &got);
let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON");
let v = hjson(pack, "vectors.json");
let out = export_pack(e, &day, v["source"].as_str().unwrap());
let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 };
hassert_same_text(pack, "vectors.json", file("vectors.json"));
hassert_same_text(pack, "vectors.h", file("vectors.h"));
assert_eq!(out.files.len(), 12);
// One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill
// kernel is present once per source that builds the table.
for (file, load, masked, hot) in [
("kernel.cu", "ds[r", " & mask]", "hot[__umulhi(r"),
("kernel_bound.cu", "ds[r", " & mask]", "hot[__umulhi(r"),
("program.metal", "dataset[r", " & MASK]", "hot[mulhi(r"),
("program_bound.metal", "dataset[r", " & MASK]", "hot[mulhi(r"),
("kernel.cl", "ds[r", " & mask]", "hot[mul_hi(r"),
] {
let text = hread(pack, file);
assert_eq!(text.matches(load).count(), dataset_loads, "{pack}/{file}: {dataset_loads} dataset loads");
assert_eq!(text.matches(masked).count(), dataset_loads, "{pack}/{file}: masked loads");
assert_eq!(text.matches(hot).count(), k, "{pack}/{file}: {k} hot loads");
assert!(text.contains(&format!("#define HOT_WORDS 0x{:08x}u", p.hot_words())), "{pack}/{file}: HOT_WORDS literal");
}
// kernel_bound.cl carries both kernels
let text = hread(pack, "kernel_bound.cl");
assert_eq!(text.matches("hot[mul_hi(r").count(), 2 * k);
assert_eq!(text.matches("ds[r").count(), 2 * dataset_loads);
for file in ["kernel.cu", "kernel.cl", "kernel_bound.cl", "memhard.metal"] {
assert_eq!(hread(pack, file).matches("igneum_hot_fill(").count(), 1, "{pack}/{file}: one hot fill kernel");
}
assert_eq!(hread(pack, "memhard.h").matches("void ht_segment(").count(), 1);
let ph = hread(pack, "program.h");
assert!(ph.contains(&format!("#define IGNEUM_HOT_MB {}", p.class.hot.unwrap().mb)));
assert!(ph.contains(&format!("#define IGNEUM_HOT_SLOTS {k}")));
assert!(ph.contains("igneum_launch_hot_fill("));
}
}

766
igneum-pow/tests/scratch.rs Normal file
View file

@ -0,0 +1,766 @@
//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes
//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results:
//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count
//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks.
//!
//! What runs under plain `cargo test`:
//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as
//! functions (question 1).
//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class
//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2).
//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to
//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16
//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the
//! hand model shown to have teeth (question 3, CPU half).
//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under
//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check;
//! the check is shown to fail on four deliberate breaks (question 4).
//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every
//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with
//! `IGNEUM_SCRATCH_PACKS_OUT=<dir>` it also writes the packs (and the edge packs) for the Metal runs of
//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape).
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel,
opencl_kernel_bound, vectors_json, LoadSource,
};
use igneum_pow::generator::{
generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT,
ITERATIONS, LANES,
};
use igneum_pow::seed::{seed_words_from_bytes, SplitMix64};
use igneum_pow::verify::{
fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource,
Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT,
};
use serde_json::Value;
use std::collections::HashMap;
use std::path::PathBuf;
/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the
/// RMW shares the readwidth branch measures.
const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"];
fn class(name: &str) -> LoadClass {
LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}"))
}
// ---------------------------------------------------------------------------------------------------------------
// 1. The written words as functions (question 1)
// ---------------------------------------------------------------------------------------------------------------
/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x`
/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform
/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`.
#[test]
fn rewrite_is_a_bijection_of_the_fold_value() {
let mut rng = SplitMix64::new(0x7363_7261_7463_6801);
for _ in 0..16 {
let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32];
let x0 = rng.next() as u32;
let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]];
for i in 0..(1u32 << 16) {
let x = x0.wrapping_add(i);
let out = scratch_rewrite(x, &w);
for j in 0..3 {
// a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are
// distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16
// bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits
let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff };
assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x");
seen[j][key as usize] = true;
}
}
}
// The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a
// rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic).
let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9];
let x = 0xdead_beef;
let out = scratch_rewrite(x, &w);
assert_eq!(out[0] ^ w[1], x);
assert_eq!((out[1] ^ w[2]).rotate_right(7), x);
assert_eq!(out[2].wrapping_sub(w[0]), x);
}
/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its
/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive
/// nonces no fill word repeats, for 8 slots x 3 words.
#[test]
fn fill_is_a_bijection_of_the_nonce() {
let seed = seed_words_from_bytes(b"igneum-genesis");
for slot in [0u32, 1, 63, 64, 255, 1023, 2047] {
for j in 0..3u32 {
let mut words: Vec<u32> = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect();
words.sort_unstable();
words.dedup();
assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct");
}
}
// base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l
assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2));
// and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0
assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2));
// the three word positions of one slot and nonce are three different permutation outputs
let f: Vec<u32> = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect();
assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]);
}
// ---------------------------------------------------------------------------------------------------------------
// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2)
// ---------------------------------------------------------------------------------------------------------------
/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots.
fn expected_distinct(s: usize, n: usize) -> f64 {
let s = s as f64;
s * (1.0 - (1.0 - 1.0 / s).powi(n as i32))
}
struct ClassStats {
units: usize,
events: usize,
hits: usize,
/// ones count per bit of the written words, 3 x 32
ones: [[u64; 32]; 3],
/// ones count per bit of written XOR read (the change the rewrite makes to the slot)
delta_ones: [[u64; 32]; 3],
/// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch)
depth: Vec<usize>,
max_depth: usize,
/// how often each slot index was addressed (the slot comes from a register's low bits)
slot_hist: Vec<u64>,
}
fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats {
let c = class(name);
let mut st = ClassStats {
units: 0,
events: 0,
hits: 0,
ones: [[0; 32]; 3],
delta_ones: [[0; 32]; 3],
depth: vec![0; 256],
max_depth: 0,
slot_hist: vec![0; c.scratch_slots_per_lane()],
};
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
for seed in seeds {
let p = generate_class(seed, c);
assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS);
for u in 0..units_per_seed {
let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000);
let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
let mut count: HashMap<(u8, u32), usize> = HashMap::new();
for e in &ev {
assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch");
let d = count.entry((e.lane, e.slot)).or_insert(0);
assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history");
assert_eq!(e.written, scratch_rewrite(e.x, &e.read));
if !e.hit {
let fill = [
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2),
];
assert_eq!(e.read, fill, "a first touch reads the fill");
}
st.depth[(*d).min(255)] += 1;
st.max_depth = st.max_depth.max(*d);
st.slot_hist[e.slot as usize] += 1;
*d += 1;
st.events += 1;
st.hits += e.hit as usize;
for j in 0..3 {
for b in 0..32 {
st.ones[j][b] += ((e.written[j] >> b) & 1) as u64;
st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64;
}
}
}
st.units += 1;
}
}
st
}
/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over
/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday
/// formula within 3 percent relative. The table printed here is the one in the analysis.
#[test]
fn written_words_unbiased_and_rehit_rates() {
let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"];
let units = 1usize << 11;
println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma");
for name in CLASSES {
let c = class(name);
let st = class_stats(name, &seeds, units);
let n = st.events as f64;
let sigma = (n / 4.0).sqrt();
let mut worst = 0.0f64;
let mut worst_delta = 0.0f64;
for j in 0..3 {
for b in 0..32 {
let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma;
let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma;
assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma");
assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma");
worst = worst.max(z);
worst_delta = worst_delta.max(zd);
}
}
let per_lane_hash = c.scratch_slots() * ITERATIONS;
let s = c.scratch_slots_per_lane();
let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash);
let exp_pct = 100.0 * exp_hits / per_lane_hash as f64;
let got_pct = 100.0 * st.hits as f64 / st.events as f64;
// chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and
// the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2)
let expect_per_slot = n / s as f64;
let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum();
let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt();
let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot;
let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot;
println!(
"{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}",
st.events, st.hits, st.max_depth
);
// a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it
assert!(
got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct,
"{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%"
);
// depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit
let shown: Vec<String> = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect();
println!(" depth histogram {}", shown.join(" "));
}
}
// ---------------------------------------------------------------------------------------------------------------
// 3. Hand-built edge programs against an independent hand model (question 3, CPU half)
// ---------------------------------------------------------------------------------------------------------------
fn ins(op: Op, dst: u8, src: u8) -> Instr {
Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
fn add_imm(dst: u8, src: u8, imm: u32) -> Instr {
Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are
/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a
/// register as the Swift edge set does.
fn edge(name: &str, c: LoadClass, instrs: Vec<Instr>) -> Program {
let seed_string = format!("igneum-scratch-edge/{name}");
let seed_bytes = seed_string.as_bytes().to_vec();
let k = instrs.iter().filter(|i| i.op == Op::Scratch).count();
assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count");
Program {
seed: seed_words_from_bytes(&seed_bytes),
seed_string,
seed_bytes,
generator: GENERATOR_VERSION,
attempt: 0,
class: c,
era_bytes: None,
instrs,
}
}
/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program).
fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> {
let m = LoadClass::scratch(1, kb).scratch_slot_mask();
let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3];
let scr = |n: usize, src: u8| -> Vec<Instr> { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() };
let mut v = Vec::new();
// every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(8, 1));
v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the in-range register MASK
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)];
p.extend(scr(8, 1));
v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the out-of-range register 0xffffffff
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)];
p.extend(scr(8, 1));
v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot 0 through the out-of-range register MASK + 1
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))];
p.extend(scr(8, 1));
v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p)));
// 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(16, 1));
v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p)));
// one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing
// (r5 is the slot register and is never a destination here)
let p: Vec<Instr> = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect();
v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p)));
// alternating slot 0 and slot MASK inside one iteration
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)];
for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() {
// r1 and r2 hold the two slots and are never destinations
p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 }));
}
v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p)));
v
}
/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its
/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth.
fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] {
let seed = &p.seed;
let m = p.class.scratch_slot_mask();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new();
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &p.instrs {
let (d, a) = (ins.dst as usize, ins.src as usize);
match ins.op {
Op::Sub => {
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]);
}
}
Op::Add => {
for lane in 0..LANES {
let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm };
r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c);
}
}
Op::Scratch => {
for lane in 0..LANES {
let slot = r[a][lane] & m;
let w = *store.entry((lane, slot)).or_insert_with(|| {
let mut f = [0u32; 3];
for j in 0..3u32 {
// the fill, written out in full rather than through verify::scratch_fill
let n = base.wrapping_add(lane as u32);
f[j as usize] = splitmix32(
(n ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E37_79B1))
.wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)),
);
}
f
});
let mut x = r[d][lane] ^ w[0];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2];
r[d][lane] = x;
let out = if mutate {
[x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])]
} else {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
};
store.insert((lane, slot), out);
}
}
other => panic!("the hand model does not implement {other:?}"),
}
}
}
let mut out = [0u64; 32];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
out[lane] = ((hi as u64) << 32) | lo as u64;
}
out
}
/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent
/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32.
const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0];
#[test]
fn edge_programs_match_the_hand_model() {
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
let mut cases = 0;
for kb in [32u8, 128] {
for (name, what, p) in edge_programs(kb) {
let slots = p.class.scratch_slots_per_lane();
for base in EDGE_BASES {
let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
let hand = hand_model(&p, base, false);
assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})");
assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth");
// the slots the trace saw are the ones the program was built to drive
let slot_set: std::collections::BTreeSet<u32> = ev.iter().map(|e| e.slot).collect();
let m = (slots - 1) as u32;
match name.as_str() {
"slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0]),
"slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![m]),
"twoslots" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0, m]),
"lanevar" => {
for e in &ev {
assert!(e.slot <= m);
}
}
_ => unreachable!(),
}
// the chain depth on the driven slot: every RMW after the first per lane is a re-hit
let per_lane = p.scratch_ops_per_hash();
let hits = ev.iter().filter(|e| e.hit).count();
let expected_hits = match name.as_str() {
"twoslots" => (per_lane - 2) * LANES,
_ => (per_lane - 1) * LANES,
};
assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits");
cases += 1;
}
}
}
assert_eq!(cases, 2 * 7 * 4);
}
// ---------------------------------------------------------------------------------------------------------------
// 4. The static scratch check over every emitted kernel of every scr pack (question 4)
// ---------------------------------------------------------------------------------------------------------------
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Dialect {
Metal,
Cuda,
OpenCl,
}
/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the
/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else
/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check:
/// the guarantee is that the emitter has one template and it masks.
pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> {
assert!(kernels >= 1);
// every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound
let k = k * kernels;
assert!(slots.is_power_of_two() && slots >= 1);
let mask = (slots - 1) as u32;
let wpl = slots * 4;
let (u, load, store, ptr) = match dialect {
Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"),
Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"),
Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"),
};
let count = |needle: &str| text.matches(needle).count();
let mut errs = Vec::new();
let mut expect = |what: &str, got: usize, want: usize| {
if got != want {
errs.push(format!("{what}: {got}, expected {want}"));
}
};
// k slot computations, each masked with exactly the class's mask and immediately followed by the one load form
expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k);
expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k);
expect("stores of the tagged slot", count(store), k);
expect("tag compares", count("(v_.x == tag)"), k);
expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k);
// the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW)
expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels);
expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k);
expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels);
expect("direct scratch indexing", count("scratch["), 0);
expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels);
// no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_`
let any_mask_load = count(&format!("u; {load}"));
expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k);
if errs.is_empty() {
Ok(())
} else {
Err(errs.join("; "))
}
}
fn packs_rw_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth")
}
fn scr_packs() -> Vec<String> {
let mut v: Vec<String> = std::fs::read_dir(packs_rw_dir())
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| n.starts_with("scr"))
.collect();
v.sort();
v
}
fn read_pack(pack: &str, file: &str) -> String {
let p = packs_rw_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel
/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot
/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena
/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first.
#[test]
fn scr_packs_regenerate_and_pass_the_static_scratch_check() {
let packs = scr_packs();
assert!(packs.len() >= 6, "the scr packs: {packs:?}");
let mut checked = 0;
let mut sample_metal = String::new();
let mut sample_cl = String::new();
let mut sample_cu = String::new();
let mut sample_k = 0;
let mut sample_slots = 0;
for pack in &packs {
let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap();
let name = j["load_class"].as_str().unwrap();
let c = class(name);
assert_eq!(&format!("{name}"), pack, "pack directory named after its class");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap();
let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap();
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let program = generate_from_seed_bytes_class(seed, &seed_bytes, c);
assert_eq!(program.class, c);
assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap());
let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
let e = Epoch { program, dataset };
let p = &e.program;
let mp = e.dataset.memhard().map(|m| &m.params);
let k = c.scratch_slots();
let slots = c.scratch_slots_per_lane();
assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS);
for (file, text, dialect, kernels) in [
("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1),
("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1),
("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1),
("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1),
("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1),
// the OpenCL bound file carries igneum_hash and igneum_hash_bound
("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2),
] {
let on_disk = read_pack(pack, file);
assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text");
// scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0
scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"));
checked += 1;
}
// the vectors of the pack are the CPU's
let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap();
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let got = e.hash_warp(base);
for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() {
let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}");
}
}
if k == 4 && slots == 64 {
sample_metal = read_pack(pack, "program.metal");
sample_cl = read_pack(pack, "kernel.cl");
sample_cu = read_pack(pack, "kernel.cu");
sample_k = k;
sample_slots = slots;
}
}
assert_eq!(checked, packs.len() * 6);
println!("static scratch check: {checked} kernels over {} scr packs", packs.len());
// The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken
// case). Each must be caught; the message names what.
assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set");
let mask = format!(" & {}u; uint4 v_", sample_slots - 1);
let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1);
assert_ne!(broken_mask, sample_metal);
let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 1 (one mask dropped, Metal): {e}");
let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;");
let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}");
println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}");
let wrong_stride = sample_metal.replace("* 256u;", "* 128u;");
let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena definition: 0, expected 1"), "{e}");
println!("break 3 (arena stride 256 -> 128 words, Metal): {e}");
let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n");
let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}");
println!("break 4 (a stray arena and scratch access, Metal): {e}");
let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 5 (one mask dropped, OpenCL): {e}");
let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 6 (one mask dropped, CUDA): {e}");
// and the unbroken texts pass under the same calls
scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap();
scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap();
scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap();
// a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class)
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err());
}
// ---------------------------------------------------------------------------------------------------------------
// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit
// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4)
// ---------------------------------------------------------------------------------------------------------------
/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases.
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
outs
}
fn contract(p: &Program) {
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots());
assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op");
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
}
#[test]
fn fuzz_scr_programs_cpu() {
let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from);
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s"
let day = "2026-10-03";
let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28);
// memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n");
let mut per_class: HashMap<String, usize> = HashMap::new();
let mut units = 0usize;
let mut wraps = 0usize;
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
// the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases
for kb in [32u8, 128] {
for (name, _what, p) in edge_programs(kb) {
let log2 = 24;
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("edge-{name}-k{kb}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge");
manifest.push_str(&format!(
"{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.class.name(),
e.program.program_id(),
e.program.scratch_ops_per_hash(),
EDGE_BASES.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
}
}
}
for i in 0..n {
let name = CLASSES[rng.below(CLASSES.len() as u64) as usize];
let c = class(name);
let seed = format!("igneum-scratch-fuzz/{i}");
let p = generate_class(&seed, c);
contract(&p);
*per_class.entry(name.to_string()).or_insert(0) += 1;
// four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in
// the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
// an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent
// kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
// the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots
for &b in &bases {
let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true);
let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0;
assert_eq!(r1.hashes, r2.hashes);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32));
units += 1;
}
if let Some(dir) = &out {
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("fuzz-{i:03}-{name}-l{log2}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.program_id(),
e.program.scratch_ops_per_hash(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
} else {
let _ = rng.below(3);
}
}
let mut classes: Vec<_> = per_class.iter().collect();
classes.sort();
println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}");
assert_eq!(units, 4 * n);
assert_eq!(wraps, n, "every program has a unit in the top 256 nonces");
if let Some(dir) = &out {
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill
/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold
/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the
/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot.
#[test]
fn slot_is_replayable_from_its_fold_values() {
let seed = seed_words_from_bytes(b"igneum-genesis");
let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32);
let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)];
let mut rng = SplitMix64::new(99);
let dsts: Vec<u32> = (0..64).map(|_| rng.next() as u32).collect();
// the honest sequence: read, fold, rewrite, 64 times
let mut w = fill;
let mut xs = Vec::new();
for &d in &dsts {
let x = fold_words(d, &w);
xs.push(x);
w = scratch_rewrite(x, &w);
}
// the replay: from the fill and the stored fold values alone
let mut w2 = fill;
for &x in &xs {
w2 = scratch_rewrite(x, &w2);
}
assert_eq!(w, w2);
// and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on
// every earlier fold value (drop one and the chain diverges)
let mut w3 = fill;
for (i, &x) in xs.iter().enumerate() {
if i != 10 {
w3 = scratch_rewrite(x, &w3);
}
}
assert_ne!(w, w3);
}

View file

@ -24,6 +24,12 @@ simulator takes the same file: `igneum-harness-sim --override-params-file infra/
Proof script: `node infra/fast-time/simnet.mjs` (3 nodes on ports 29500 and up, `igneum-devnet-950`, data
`/tmp/igneum-fast-time`; records the first lock and the first program swap; see "Measured" below).
Class switch gate (Counter ASIC 2.0 rollout G4): `node infra/fast-time/class-v3.mjs` (3 nodes on ports 29600 and up,
`igneum-devnet-960`, data `/tmp/igneum-fast-time-v3`, CPU difficulty `genesis_bits` 0x1f010000, real CPU mining on
every node, `program_class_v3_activation_daa` a few epochs ahead; reports blocks on both sides of the boundary, the
program id and class before and after, rejected blocks, forks, and every node's switch line; see
`docs/plans/counter-asic-2-node.md`).
Miners: `igneum-miner` follows the epoch length and lead its node reports in every template (`pow_epoch`), no flag.
The dataset day is not in the template, so a real-hash miner on a fast-time network takes `IGNEUM_POW_DAY_MS=1440000`
in its environment (the node reads it from the file). The environment variables `IGNEUM_POW_EPOCH_BLOCKS`,
@ -51,6 +57,11 @@ Time parameters, divided by 60 (devnet value, 60x value):
| `difficulty_v2_activation_daa` | never (`18446744073709551615`) | never | a height, not a clock: difficulty rule v2 (4 Oct 2026, `docs/analysis/difficulty-2026-10-04-oscillation.md`) switches on at this DAA score; a test network sets it in its merged file (`sim/difficulty/testnet_v2.py` uses 900) |
| `proving_v0_activation_daa` | never | never | a height: proving v0 payouts (spec 7.7) switch on at this DAA score; `tools/proving-v0/run.mjs` sets 60 in its merged file |
| `finality_v3_activation_daa` | never | never | a height, not a clock: finality rule v3 (the frozen weight table of ledger F21 and the certificate fold of F22, 4 Oct 2026 evening) applies to checkpoints at or above this DAA score; `tools/finality-attacks/v3.mjs` sets 0 in its merged file |
| `pow_genesis_dataset_log2` | 28 | 28 | a size, not a clock: the dataset at genesis (2^28 words, 1 GiB); the cache growth rule (ca2-mixer) doubles it with the days since the genesis day |
| `proving_v1_activation_daa` | never | never | a height: proving v1 (the fork's proving-v1 branch, 5 Oct 2026) switches on at this DAA score |
| `proving_v1_segment_blocks`, `proving_v1_aggregator_share_bps` | 8, 1,000 | 8, 1,000 | a block count and a share |
| `proving_v1_unproven_daa` | 600 | 10 | a DAA clock (the unproven allowance), divided by 60; the proving agent confirms the value |
| `program_class_v3_activation_daa` | never | never | a height, not a clock: the lottery hash draws programs from class v3 (Counter ASIC 2.0, 5 Oct 2026) from the first EPOCH whose start is at or above this DAA score (rounded up to an epoch boundary: at 60 DAA per epoch, 150 means epoch 3 at DAA 180); `infra/fast-time/class-v3.mjs` sets it a few epochs ahead in its merged file |
Unchanged, and why:

View file

@ -0,0 +1,282 @@
#!/usr/bin/env node
// Counter ASIC 2.0 rollout gate G4 (docs/plans/counter-asic-2-rollout.md section 7): the fast-time 3-node network
// mining across a program class v3 activation. A private network on 127.0.0.1 ports 29600 and up, data under
// /tmp/igneum-fast-time-v3, network id igneum-devnet-960, every node on infra/fast-time/override-60x.json merged with
// a CPU genesis difficulty (genesis_bits 0x1f010000, 2^16 hashes per block, as sim/difficulty/testnet_v2.py) and
// `program_class_v3_activation_daa` a few epochs ahead (default 150: inside epoch 2 at 60 DAA per epoch, so the
// switch rounds UP to epoch 3 at DAA 180, which is the boundary rule under test). One real CPU miner per node
// (igneum-miner --engine igneum-pow, 1 thread) follows its node's templates, so every block of the run is a real
// lottery-hash solution and every node verifies every block of the other two under the class of its epoch.
//
// Reports, from the nodes' RPC and the logs: blocks on each side of the boundary, the program class and id of every
// epoch (before and after), rejected blocks (the miners' submit answers and the nodes' "PoW rejected" lines), forks
// (every node's sink, selected tip and block count at the end), and every node's switch line. Exit 0 when every
// check passes. Never touches the live devnet (26610/26611, 26640/26641, 28640) or the simnet.mjs ports.
//
// node infra/fast-time/class-v3.mjs [--secs 420] [--activation 150] [--epochs-after 2] [--threads 1]
// [--metal <igneum-bench>] [--genesis-bits 0x1f010000]
// IGNEUMD and IGNEUM_MINER name the binaries (default: the ca2 fork worktree's target-ca2/release).
// --metal: node 0's miner is a real Metal GPU worker (igneum-miner --worker <igneum-bench> --prepare-packs <dir>
// --exit-on-seed-change, the app's own shape), so the Metal worker's class v3 path (the prepare line with the pack
// directory and the class and era tokens, the pack-built program and day) mines across the boundary (gate G4b);
// the report then adds the miner's PREPARE lines, the worker's prepared lines, any need / mismatch / refusal line,
// the accepted blocks on each side and the swap time. --genesis-bits raises the CPU difficulty for a GPU miner
// (0x1e010000 = 2^24 hashes per block: under a second on an M5 Max; the CPU miners on nodes 1 and 2 then verify
// and rarely find).
import { spawn } from 'node:child_process';
import { mkdirSync, rmSync, writeFileSync, readFileSync, openSync, existsSync } from 'node:fs';
import { connectRpc } from '../../tools/finality-attacks/lib/rpc.mjs';
import { devAddress } from '../../tools/harness/lib/address.mjs';
const ROOT = new URL('../../', import.meta.url).pathname;
const FILE = `${ROOT}infra/fast-time/override-60x.json`;
const BIN = process.env.IGNEUM_CA2_BIN || `${ROOT}vendor/igneum-node-ca2/target-ca2/release`;
const IGNEUMD = process.env.IGNEUMD || `${BIN}/igneumd`;
const CPU_MINER = process.env.IGNEUM_MINER || `${BIN}/igneum-miner`;
const TMP = '/tmp/igneum-fast-time-v3';
const BASE = 29600, SUFFIX = 960;
const args = process.argv.slice(2);
const flag = (name, dflt) => { const i = args.indexOf(`--${name}`); return i >= 0 ? +args[i + 1] : dflt; };
const sflag = (name) => { const i = args.indexOf(`--${name}`); return i >= 0 ? args[i + 1] : null; };
const GENESIS_BITS = flag('genesis-bits', 0x1f010000);
const METAL = sflag('metal');
const SECS = flag('secs', 420);
const ACTIVATION = flag('activation', 150);
const EPOCHS_AFTER = flag('epochs-after', 2);
const THREADS = flag('threads', 1);
const started = [];
const log = (...a) => console.log(new Date().toISOString().slice(11, 23), ...a);
const sleep = (ms) => new Promise(r => setTimeout(r, ms));
for (const b of [IGNEUMD, CPU_MINER, ...(METAL ? [METAL] : [])]) if (!existsSync(b)) { console.error(`missing ${b}`); process.exit(2); }
rmSync(TMP, { recursive: true, force: true }); mkdirSync(TMP, { recursive: true });
// The file is merged as TEXT, never through JSON.parse: a `never` height is 18446744073709551615, which JavaScript
// rounds to 1.8446744073709552e+19 and the node refuses ("expected u64"; the first gate run, 5 October 2026 21:30Z).
// The merged fields are appended after the file's last field; a field already in the file is removed first.
const baseText = readFileSync(FILE, 'utf8');
const field = (name) => { const m = new RegExp(`"${name}":\\s*([0-9]+)`).exec(baseText); return m ? +m[1] : undefined; };
const EPOCH = field('pow_epoch_blocks');
const DAY_MS = field('pow_day_ms');
const FIRST_V3_EPOCH = Math.ceil(ACTIVATION / EPOCH);
const BOUNDARY = FIRST_V3_EPOCH * EPOCH;
const override = `${TMP}/override.json`;
export function mergeOverrideText(text, fields) {
let out = text;
for (const k of Object.keys(fields)) out = out.replace(new RegExp(`\\s*"${k}":\\s*[^,}\\n]+,?`), '');
const extra = Object.entries(fields).map(([k, v]) => `"${k}": ${typeof v === 'string' ? JSON.stringify(v) : v}`).join(', ');
return out.replace(/,?\s*}\s*$/, `,\n ${extra}\n}\n`);
}
writeFileSync(override, mergeOverrideText(baseText, { genesis_bits: GENESIS_BITS, skip_proof_of_work: false, program_class_v3_activation_daa: ACTIVATION }));
log(`activation ${ACTIVATION} at ${EPOCH} DAA per epoch: the first v3 epoch is ${FIRST_V3_EPOCH} (DAA ${BOUNDARY}); run ${SECS} s, CPU genesis bits 0x${GENESIS_BITS.toString(16)}`);
class Node {
constructor(i, connect = []) {
this.i = i; this.grpcPort = BASE + i * 10; this.p2pPort = BASE + i * 10 + 1; this.jsonPort = BASE + i * 10 + 2;
this.connect = connect; this.dir = `${TMP}/n${i}`; this.logFile = `${this.dir}/node.log`;
}
get grpc() { return `grpc://127.0.0.1:${this.grpcPort}`; }
async start() {
mkdirSync(this.dir, { recursive: true });
const a = ['--devnet', `--devnet-suffix=${SUFFIX}`, '--nodnsseed', '--disable-upnp', '--nologfiles', '--enable-unsynced-mining', '--utxoindex',
`--appdir=${this.dir}`, `--rpclisten=127.0.0.1:${this.grpcPort}`, `--rpclisten-json=127.0.0.1:${this.jsonPort}`,
`--listen=127.0.0.1:${this.p2pPort}`, `--override-params-file=${override}`, '--loglevel=info', '--yes'];
if (this.connect.length) a.push(`--connect=${this.connect.join(',')}`); else a.push('--outpeers=0');
const out = openSync(this.logFile, 'a');
this.proc = spawn(IGNEUMD, a, { stdio: ['ignore', out, out] });
started.push(this.proc);
await sleep(1200);
this.rpc = await connectRpc(`ws://127.0.0.1:${this.jsonPort}`);
log(`n${this.i} up pid ${this.proc.pid} json ${this.jsonPort} p2p ${this.p2pPort}`);
return this;
}
grepLog(re) { try { return readFileSync(this.logFile, 'utf8').split('\n').filter(l => re.test(l)); } catch { return []; } }
}
function miner(bin, argv, name, env = {}) {
const out = openSync(`${TMP}/${name}.log`, 'a');
const p = spawn(bin, argv, { stdio: ['ignore', out, out], env: { ...process.env, ...env } });
started.push(p);
return p;
}
async function stopAll() {
for (const p of started.reverse()) { try { p.kill('SIGINT'); } catch { } }
await sleep(1500);
for (const p of started) { try { p.kill('SIGKILL'); } catch { } }
}
process.on('SIGINT', async () => { await stopAll(); process.exit(130); });
process.on('unhandledRejection', async (e) => { log(`FAILED: ${e?.stack || e}`); await stopAll(); process.exit(3); }); // a thrown start leaves no node behind
const minerName = (i) => (METAL && i === 0) ? 'metal0' : `cpu${i}`;
const minerLog = (i) => { try { return readFileSync(`${TMP}/${minerName(i)}.log`, 'utf8').split('\n'); } catch { return []; } };
const t0 = Date.now();
const since = () => ((Date.now() - t0) / 1000).toFixed(1);
const n0 = await new Node(0).start();
const n1 = await new Node(1, [`127.0.0.1:${n0.p2pPort}`]).start();
const n2 = await new Node(2, [`127.0.0.1:${n0.p2pPort}`]).start(); // --connect takes one address
const nodes = [n0, n1, n2];
for (const n of nodes) {
const line = n.grepLog(/Program class v3 from the override file/)[0];
log(`n${n.i} switch line: ${line ? line.replace(/^.*?(Program class v3)/, '$1') : '(none)'}`);
}
log(`n0 PoW schedule: ${n0.grepLog(/PoW schedule from the override file/).map(l => l.replace(/^.*?(PoW schedule)/, '$1')).join(' | ') || '(no line)'}`);
log(`n0 digest: ${n0.grepLog(/Consensus params digest/).map(l => l.replace(/^.*?digest: /, '').slice(0, 16)).join(' ')}`);
// one real CPU miner per node, 1 thread, 2^16 expected hashes per block at genesis; with --metal, node 0's miner drives
// the Metal worker the way the app does (the pack for every prepared pair under packs/prepare, exit 42/44 on a refusal)
const PACKS = `${TMP}/packs/prepare`;
mkdirSync(PACKS, { recursive: true });
nodes.forEach((n, i) => {
if (METAL && i === 0) {
miner(CPU_MINER, ['mine', n.grpc, '1', String(SECS), 'metal0', '--engine', 'igneum-pow', '--worker', METAL, '--prepare-packs', PACKS, '--exit-on-seed-change', '--payout-label', 'metal0', '--status-secs', '30', '--no-vote'], 'metal0', { IGNEUM_POW_DAY_MS: String(DAY_MS) });
log(`n0 miner: Metal worker ${METAL}, packs under ${PACKS}`);
} else {
miner(CPU_MINER, ['mine', n.grpc, String(THREADS), String(SECS), `cpu${i}`, '--engine', 'igneum-pow', '--payout-label', `cpu${i}`, '--status-secs', '30', '--no-vote'], `cpu${i}`, { IGNEUM_POW_DAY_MS: String(DAY_MS) });
}
});
const pay = devAddress('fast-time-v3');
const epochs = new Map(); // epoch index -> { class, firstSeenDaa, at }
let firstV3 = null, lastEpoch = -1, lastReport = 0, lastDaa = 0, switchEnd = null;
const samples = [];
while (Date.now() - t0 < SECS * 1000) {
await sleep(1000);
let daa = null, epoch = null, cls = null, nextCls = null, era = null, eraSeed = null, act = null;
try {
const t = await n0.rpc.call('getBlockTemplate', { payAddress: pay, extraData: [] });
const pe = t.powEpoch || t.pow_epoch || {};
daa = pe.virtualDaaScore ?? t.block?.header?.daaScore; epoch = pe.epochIndex; cls = pe.programClass; nextCls = pe.nextProgramClass;
era = pe.eraIndex; eraSeed = pe.eraSeed; act = pe.programClassV3ActivationDaa;
} catch (e) { log(`template: ${e.message}`); }
if (epoch != null && epoch !== lastEpoch) {
epochs.set(epoch, { class: cls, firstSeenDaa: daa, at: +since() });
log(`epoch ${lastEpoch} -> ${epoch} at daa ${daa}, ${since()} s: template class ${cls} (generator), next epoch class ${nextCls}, era ${era} seed ${String(eraSeed).slice(0, 16)}, activation ${act}`);
if (firstV3 == null && cls === 3) { firstV3 = { epoch, daa, at: +since() }; log(`CLASS SWITCH: the template is class v3 from epoch ${epoch} (daa ${daa}) at ${since()} s wall`); }
lastEpoch = epoch;
}
lastDaa = daa ?? lastDaa;
if (Date.now() - lastReport > 15000) {
lastReport = Date.now();
const counts = await Promise.all(nodes.map(async n => { try { const d = await n.rpc.call('getBlockDagInfo'); return `${d.blockCount}/${String(d.sink).slice(0, 8)}`; } catch { return '?'; } }));
log(`t=${since()} s daa ${daa} epoch ${epoch} class ${cls} blocks/sink per node ${counts.join(' ')}`);
samples.push({ t: +since(), daa, epoch, class: cls, nodes: counts });
}
if (firstV3 != null && daa != null && daa >= BOUNDARY + EPOCHS_AFTER * EPOCH) { switchEnd = +since(); break; }
}
await sleep(3000); // let the last blocks relay before the end-of-run reads
// ---- the end-of-run reads -----------------------------------------------------------------------------------------
const dag = await Promise.all(nodes.map(async n => { try { return await n.rpc.call('getBlockDagInfo'); } catch (e) { return { error: e.message }; } }));
const genesis = dag[0].pruningPointHash;
// every block the first node holds, by DAA score, from genesis
async function allBlocks(n) {
const out = []; let low = genesis; const seen = new Set();
for (let round = 0; round < 500; round++) {
const r = await n.rpc.call('getBlocks', { lowHash: low, includeBlocks: true, includeTransactions: false });
const blocks = r.blocks || [];
let added = 0;
for (const b of blocks) { const h = b.verboseData?.hash || b.header?.hash; if (seen.has(h)) continue; seen.add(h); out.push({ hash: h, daa: +b.header.daaScore, chain: !!b.verboseData?.isChainBlock, blue: +(b.verboseData?.blueScore ?? 0) }); added++; }
if (!blocks.length || added === 0) break;
low = (r.blockHashes || []).at(-1) || blocks.at(-1).verboseData?.hash; if (!low) break;
}
return out;
}
let blocks = [];
try { blocks = await allBlocks(n0); } catch (e) { log(`getBlocks: ${e.message}`); }
const before = blocks.filter(b => b.daa < BOUNDARY), after = blocks.filter(b => b.daa >= BOUNDARY);
const chainBefore = before.filter(b => b.chain).length, chainAfter = after.filter(b => b.chain).length;
// program ids and classes per epoch from the miners' "program and 256 MiB cache ready" lines
const programs = new Map(); // epoch seed -> { class, id, miners: Set }
for (const i of [0, 1, 2]) for (const l of minerLog(i)) {
const m = /epoch seed ([0-9a-f]{64}) day (\d+) \(daa (\d+)\): program and 256 MiB cache ready in ([\d.]+) ms; class (v\d) program id ([0-9a-f]{16})/.exec(l);
if (!m) continue;
const k = m[1]; const e = programs.get(k) || { seed: k.slice(0, 16), epoch: Math.floor(+m[3] / EPOCH), class: m[5], id: m[6], miners: new Set(), ready_ms: [] };
if (e.id !== m[6] || e.class !== m[5]) e.disagree = true;
e.miners.add(i); e.ready_ms.push(+m[4]); programs.set(k, e);
}
// ready_ms: the program generation plus the day's cache and dataset build on one CPU core (the first epoch of a day
// pays the cache; under the mixer x4 construction the first v3 epoch's build is the number the rollout asks for)
const programRows = [...programs.values()].sort((a, b) => a.epoch - b.epoch).map(p => ({ epoch: p.epoch, class: p.class, program_id: p.id, seed: p.seed, miners: p.miners.size, disagree: !!p.disagree, ready_ms: p.ready_ms }));
const v2Ids = programRows.filter(p => p.class === 'v2').map(p => p.program_id), v3Ids = programRows.filter(p => p.class === 'v3').map(p => p.program_id);
// rejections: the miners' submit answers and the nodes' PoW lines
const accepted = [0, 1, 2].map(i => minerLog(i).filter(l => /ACCEPTED block/.test(l)).length);
const rejectedMiner = [0, 1, 2].map(i => minerLog(i).filter(l => /rejected nonce=|submit error/.test(l)));
const rejectedNode = nodes.map(n => n.grepLog(/PoW rejected|Rejected block|rejected block/i));
const switchLines = nodes.map(n => n.grepLog(/Program class v3 from the override file/).map(l => l.replace(/^.*?(Program class v3)/, '$1'))[0] || null);
const sinks = dag.map(d => String(d.sink || '?').slice(0, 16));
const counts = dag.map(d => d.blockCount ?? '?');
const tips = dag.map(d => (d.tipHashes || []).length);
// the Metal miner's protocol lines (gate G4b): PREPARE sent (the miner), prepared / need / error (the worker), the swap
let metal = null;
if (METAL) {
const L = minerLog(0);
const t = (l) => { const m = /^(\d+\.\d+) /.exec(l); return m ? +m[1] * 1000 : null; };
const switchAt = firstV3 ? t0 + firstV3.at * 1000 : null;
const prepares = L.filter(l => /PREPARE sent for epoch seed/.test(l)).map(l => l.replace(/^\S+ /, ''));
const prepared = L.filter(l => /worker: prepared /.test(l)).map(l => l.replace(/^\S+ /, ''));
const preparedV3 = prepared.filter(l => / class v3 /.test(l));
const need = L.filter(l => /worker: need |^\S+ worker error:.*need /.test(l) || /\bneed [0-9a-f]{64}/.test(l));
// a PACK OUT OF DATE before the first class v3 prepare is the v2 rebuild path at work (a seed that flipped inside the
// quarter-lead confirm window, 3 DAA on the 60x profile); the gate judges the v3 path from its first prepare on
const firstV3Prepare = L.findIndex(l => /PREPARE sent for epoch seed .* class v3/.test(l));
const sinceV3 = (l, i) => firstV3Prepare < 0 || i >= firstV3Prepare;
const mismatch = L.filter((l, i) => sinceV3(l, i) && /program class mismatch|era seed mismatch|PACK OUT OF DATE|pack .*: program pack/.test(l));
const mismatchBefore = L.filter((l, i) => !sinceV3(l, i) && /program class mismatch|era seed mismatch|PACK OUT OF DATE/.test(l));
const refused = L.filter(l => /exiting with code 4[24]|refused the program pack|prepare-failed/.test(l));
const swaps = L.filter(l => /swapped with no pause|compiles inline|worker without prepare support/.test(l)).map(l => l.replace(/^\S+ /, ''));
const accepted = L.filter(l => /ACCEPTED block/.test(l));
const acceptedAfter = switchAt ? accepted.filter(l => (t(l) || 0) >= switchAt).length : 0;
const found = L.filter(l => /worker: found |^\S+ found /.test(l)).length;
const status = L.filter(l => /miner 'metal0' \[igneum-pow\]/.test(l)).map(l => l.replace(/^\S+ /, ''));
const lastStatus = status.at(-1) || '';
const mismatched = +(/mismatched=(\d+)/.exec(lastStatus)?.[1] ?? 0);
const rate = /hash=([\d.]+) MH\/s/.exec(lastStatus)?.[1];
metal = { worker: METAL, prepares, prepared, prepared_v3: preparedV3, need: need.length, mismatch_lines: mismatch, v2_rebuild_lines_before_v3: mismatchBefore, refused, swaps, accepted_total: accepted.length, accepted_after_switch: acceptedAfter, found_lines: found, cpu_recheck_mismatched: mismatched, rate_mh_s: rate, last_status: lastStatus };
}
const checks = {
switch_line_on_every_node: switchLines.every(Boolean),
switch_line_names_the_rounded_epoch: switchLines.every(l => l && l.includes(`active from epoch ${FIRST_V3_EPOCH} `)),
template_switched_at_the_first_v3_epoch: firstV3 != null && firstV3.epoch === FIRST_V3_EPOCH,
blocks_before_the_boundary: before.length > 0,
blocks_after_the_boundary: after.length > 0,
v2_and_v3_programs_seen: v2Ids.length > 0 && v3Ids.length > 0,
program_ids_differ_across_the_switch: v2Ids.length > 0 && v3Ids.length > 0 && !v2Ids.some(id => v3Ids.includes(id)),
miners_agree_on_every_program: programRows.every(p => !p.disagree),
zero_rejected_by_miners: rejectedMiner.every(r => r.length === 0),
zero_rejected_by_nodes: rejectedNode.every(r => r.length === 0),
sinks_agree: new Set(sinks).size === 1,
block_counts_agree: new Set(counts.map(String)).size === 1,
...(METAL ? {
metal_prepare_sent_for_v3: metal.prepares.some(l => / class v3 /.test(l)),
metal_worker_prepared_v3_pack: metal.prepared_v3.length > 0,
metal_no_need_or_mismatch: metal.need === 0 && metal.mismatch_lines.length === 0 && metal.refused.length === 0,
metal_accepted_blocks_after_switch: metal.accepted_after_switch > 0,
metal_cpu_recheck_clean: metal.cpu_recheck_mismatched === 0,
metal_swapped_without_pause: metal.swaps.some(l => /swapped with no pause/.test(l)),
} : {}),
};
const pass = Object.values(checks).every(Boolean);
const summary = {
pass, checks, activation: ACTIVATION, epoch_blocks: EPOCH, first_v3_epoch: FIRST_V3_EPOCH, boundary_daa: BOUNDARY, secs: SECS, threads: THREADS,
node: IGNEUMD, miner: CPU_MINER, genesis_bits: `0x${GENESIS_BITS.toString(16)}`,
template_switch: firstV3, run_ended_at_s: switchEnd, final_daa: lastDaa,
blocks: { total: blocks.length, before_boundary: before.length, after_boundary: after.length, chain_before: chainBefore, chain_after: chainAfter },
programs: programRows, accepted_per_miner: accepted,
rejected_by_miners: rejectedMiner.map(r => r.length), rejected_by_nodes: rejectedNode.map(r => r.length),
rejected_lines: [...rejectedMiner.flat(), ...rejectedNode.flat()].slice(0, 20),
sinks, block_counts: counts, tips_per_node: tips, switch_lines: switchLines, samples, metal,
};
writeFileSync(`${TMP}/summary.json`, JSON.stringify(summary, null, 2));
log(`SUMMARY ${pass ? 'PASS' : 'FAIL'}: blocks ${before.length} before / ${after.length} after the boundary at DAA ${BOUNDARY} (chain ${chainBefore} / ${chainAfter}); programs ${programRows.map(p => `e${p.epoch}:${p.class}:${p.program_id}:${Math.max(...p.ready_ms)}ms`).join(' ')}; rejected miners ${rejectedMiner.map(r => r.length).join('/')} nodes ${rejectedNode.map(r => r.length).join('/')}; sinks ${sinks.join(' ')} (${checks.sinks_agree ? 'agree' : 'DIFFER'}); block counts ${counts.join('/')}; switch lines ${switchLines.filter(Boolean).length}/3`);
if (metal) {
log(`METAL: ${metal.prepares.length} PREPARE lines (${metal.prepares.filter(l => / class v3 /.test(l)).length} class v3), ${metal.prepared.length} prepared (${metal.prepared_v3.length} v3), need ${metal.need}, mismatch/refusal lines ${metal.mismatch_lines.length + metal.refused.length}, accepted ${metal.accepted_total} (${metal.accepted_after_switch} after the switch), cpu re-check mismatched ${metal.cpu_recheck_mismatched}, rate ${metal.rate_mh_s} MH/s`);
for (const l of [...metal.prepares, ...metal.prepared, ...metal.swaps]) log(` ${l.slice(0, 260)}`);
for (const l of [...metal.mismatch_lines, ...metal.refused].slice(0, 10)) log(` BAD ${l.slice(0, 260)}`);
}
for (const [k, v] of Object.entries(checks)) if (!v) log(`FAILED CHECK ${k}`);
log(`summary: ${TMP}/summary.json`);
await stopAll();
process.exit(pass ? 0 : 1);

View file

@ -54,6 +54,16 @@
"difficulty_v2_activation_daa": 18446744073709551615,
"proving_v0_activation_daa": 18446744073709551615,
"finality_v3_activation_daa": 18446744073709551615,
"program_class_v3_activation_daa": 18446744073709551615,
"pow_genesis_dataset_log2": 28,
"proving_v1_activation_daa": 18446744073709551615,
"proving_v1_segment_blocks": 8,
"proving_v1_unproven_daa": 10,
"proving_v1_aggregator_share_bps": 1000,
"fees_v1_activation_daa": 0,
"proving_v1_activation_daa": 18446744073709551615,
"proving_v1_segment_blocks": 8,
"proving_v1_unproven_daa": 10,
"proving_v1_aggregator_share_bps": 1000,
"fees": {"pgas": {"version": 1, "cycles_per_pgas": 1000, "intrinsic_pgas_per_tx": 300, "modexp_base": 10, "modexp_per_byte_numer": 1, "modexp_per_byte_denom": 10}, "block_proving_gas_limit": 120000, "shard_proving_gas_budget": 30000, "min_execution_base_fee_wei": 100000000000, "min_proving_base_fee_wei": 10000000000000, "initial_execution_base_fee_wei": 100000000000, "initial_proving_base_fee_wei": 10000000000000, "base_fee_change_denominator": 8}
}

View file

@ -35,7 +35,9 @@ for (const b of [IGNEUMD, VMINE]) if (!existsSync(b)) { console.error(`missing $
rmSync(TMP, { recursive: true, force: true }); mkdirSync(TMP, { recursive: true });
const override = `${TMP}/override.json`;
writeFileSync(override, JSON.stringify({ ...JSON.parse(readFileSync(FILE, 'utf8')), skip_proof_of_work: true }));
// merged as text, not through JSON.parse: a `never` height (18446744073709551615) does not survive a JavaScript number
// (the class-v3.mjs gate run of 5 October 2026 21:30Z: "invalid type: floating point 1.8446744073709552e+19")
writeFileSync(override, readFileSync(FILE, 'utf8').replace(/\s*"skip_proof_of_work":\s*[^,}\n]+,?/, '').replace(/,?\s*}\s*$/, ',\n "skip_proof_of_work": true\n}\n'));
class Node {
constructor(i, connect = []) {

View file

@ -106,3 +106,22 @@ from the repository root (the project's root directory is `site`), but the norma
(`igneum-relay`) and the downloads folder (`igneum-dl`) are the other two projects in the `igneum` team; both deploy by CLI
from their own folders (`relay/README.md`, `packaging/ota/README.md`). CLAUDE.md still says the site sits in the vivanmn
team and deploys with `--scope vivanmn` from `site/`: that was true on 3 October and is not now.
## Job scripts (5 October 2026, the lost-quote class)
A bash body in a PowerShell job (`relay/playbooks/*.ps1`, `tools/**/*.ps1`) is written to a file and run with
`bash <file>`, never inline. That is the 0.3.6 cut's rule. Twice on 5 October a body travelled inside a PowerShell
string, a quote was lost on the way through PowerShell, bash refused the whole body, and the job either reported exit 0
having done nothing (`tools/amd-prove/pc1-cpu-prove.ps1`, first version: an apostrophe inside a single-quoted awk
program) or failed in 4 s (the 0.3.10 installer job). The file shape keeps the body readable by `bash -n` before it runs;
`pc1-cpu-prove.ps1` does that in the job itself. `tools/ci/bash-body-check.sh` fails CI on an inline body that does not
parse: it finds every body handed to bash (`bash -c "..."`, `bash -lc $var`, a `+` concatenation, a here-string written to a
file and run), unescapes it the way PowerShell would, and runs `bash -n` on it. A body it sees but cannot read fails too.
`--self-test` shows it firing on the fixtures under `tools/ci/fixtures/`. A job script is published from a worktree and
never passes CI before it runs, so `packaging/ota/publish-jobs.sh add --kind run` runs the same check (and
`tools/ci/prover-socket-check.sh`, the root-socket class; `bash -n` for a `.sh` script) on the script before anything is
signed, and refuses the publish with the check's output. `packaging/ota/test-publish-jobs.sh` proves the refusals.
The wiped-jobs-folder class (5 October 2026, 21:49Z): an app install clears the jobs folder, so a run job that uses a kit
fetched by an earlier fetch job tests the kit is there (Test-Path on the kit root or any file under it) before its first
use, and the fetch is republished under a new id after any app update. `tools/ci/kit-path-check.sh` fails CI and the
publish on a kit path used before that check.

View file

@ -11,7 +11,7 @@
<key>CFBundleVersion</key>
<string>VERSION_STAMP</string>
<key>CFBundleShortVersionString</key>
<string>0.3.10</string>
<string>0.3.11</string>
<key>CFBundlePackageType</key>
<string>APPL</string>
<key>CFBundleExecutable</key>

View file

@ -28,7 +28,7 @@ DL_HOST="https://dl.igneum.network"
# carry the devnet's activation height here, the same N as every other devnet node, before it is cut (Mac and CI alike:
# make-payload.sh sources this file). Rule and order: docs/plans/difficulty-v2-rollout-devnet.md.
# Example: NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa": 123456}'
NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}'
NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000}'
# igneum_secret_file <env var name> <base name> -> the file to read: the variable when set, else <base>.next when it
# exists, else <base>; IGNEUM_CONFIG_DIR (tests) replaces ~/.config/igneum

View file

@ -256,6 +256,22 @@ if [ "$CMD" = add ]; then
[ -n "$SCRIPT" ] && [ -f "$SCRIPT" ] || { echo "run: --script <file> is required" >&2; exit 2; }
[ -n "$SHELL_KIND" ] || { case "$SCRIPT" in *.sh) SHELL_KIND=bash ;; *) SHELL_KIND=powershell ;; esac; }
[ -n "$PLATFORM" ] || { [ "$SHELL_KIND" = powershell ] && PLATFORM=windows || PLATFORM=any; }
# A job script is published from a worktree and never passes CI before it runs (5 October 2026: the root-socket
# fault came back from a branch without the check), so the class checks run here on the script itself, before
# anything is signed. A failure refuses the publish with the check's output; a missing check refuses it too.
# PowerShell: every inline bash body must parse (tools/ci/bash-body-check.sh, the lost-quote class; a body it
# cannot read fails, never skips). Bash: the script itself must parse. Both: a root prover run kills the GPU
# server and unlinks its socket (tools/ci/prover-socket-check.sh, the root-socket class), and a fetched kit under
# the jobs folder is tested before its first use (tools/ci/kit-path-check.sh, the wiped-jobs-folder class). A
# check file that is missing from this tree refuses too (bash exits 127 with the reason), so no job goes out
# unchecked.
if [ "$SHELL_KIND" = powershell ]; then
out="$(bash "$ROOT/tools/ci/bash-body-check.sh" "$SCRIPT" 2>&1)" || { echo "run: bash-body-check.sh refuses $SCRIPT:" >&2; printf '%s\n' "$out" >&2; exit 1; }
else
out="$(bash -n "$SCRIPT" 2>&1)" || { echo "run: bash -n refuses $SCRIPT (the lost-quote class):" >&2; printf '%s\n' "$out" >&2; exit 1; }
fi
out="$(bash "$ROOT/tools/ci/prover-socket-check.sh" "$SCRIPT" 2>&1)" || { echo "run: prover-socket-check.sh refuses $SCRIPT:" >&2; printf '%s\n' "$out" >&2; exit 1; }
out="$(bash "$ROOT/tools/ci/kit-path-check.sh" "$SCRIPT" 2>&1)" || { echo "run: kit-path-check.sh refuses $SCRIPT:" >&2; printf '%s\n' "$out" >&2; exit 1; }
PARAMS="$(python3 -c 'import json,sys; print(json.dumps({"script": open(sys.argv[1]).read(), "shell": sys.argv[2], "elevated": sys.argv[3]=="1", "stop_miners_first": sys.argv[4]=="1", **({"timeout_minutes": int(sys.argv[5])} if sys.argv[5] else {})}))' "$SCRIPT" "$SHELL_KIND" "$ELEVATED" "$STOP_MINERS" "$TIMEOUT_MIN")"
[ -n "$TITLE" ] || TITLE="run $(basename "$SCRIPT")"
;;

View file

@ -8,7 +8,11 @@
# byte and its sig IS the plain signature; the signer reads the envelope back; a second `add` (a new publish) gives
# a new envelope; an envelope made by hand from the FIRST file and the SECOND signature (the stale pair an edge can
# serve across a deploy) is REFUSED by the signer with the words the app logs; a tampered inner byte is refused;
# `sign` rewrites all three consistently; and the real OTA key is the key the app embeds.
# `sign` rewrites all three consistently; a `run` script that fails a class check (a lost quote in an inline bash
# body, a body the check cannot read, a bash script that does not parse, a root prover run without the socket cleanup,
# a fetched kit used before a presence check)
# is REFUSED before anything is signed and the envelope is left as it was; and the real OTA key is the key the app
# embeds.
#
# A check is trusted only once it has fired on a known-good and a known-bad case (standing rule, 4 October 2026), so
# every negative case here must FAIL for the run to pass. Needs ~/.config/igneum/ota-signing-key (the publisher's own
@ -96,6 +100,24 @@ PY
grep -q "^inner-identical True$" "$T/inner2.txt" && grep -q "^sig-identical True$" "$T/inner2.txt" && ok "after sign: the envelope still holds the plain file and its signature" || bad "after sign: the envelope and the pair differ"
expect_ok "list reads the folder" "$PUBLISH" list --dest "$D"
echo "== the class checks refuse a bad script before anything is signed (5 October 2026: a job never passes CI first)"
cp "$D/igneum-jobs.signed.json" "$T/before.signed"
printf '& wsl.exe -d Ubuntu-24.04 -- bash -c '"'"'echo "started; ls /opt/igneum'"'"' 2>&1\n' > "$T/lost-quote.ps1"
expect_fail "a PowerShell script with a lost quote in its bash body" "$PUBLISH" add --kind run --target 1ccfe586 --script "$T/lost-quote.ps1" --id test-lost-quote --dest "$D" --base-url https://example.invalid/dl/t
grep -q "unexpected EOF while looking for matching" "$T/out" && ok "refused with the bash -n error" || bad "another reason: $(tail -1 "$T/out")"
printf '$cmd = (Get-Content body.txt) -join "; "\n& wsl.exe -- bash -c $cmd\n' > "$T/unreadable.ps1"
expect_fail "a PowerShell script whose bash body the check cannot read" "$PUBLISH" add --kind run --target 1ccfe586 --script "$T/unreadable.ps1" --id test-unreadable --dest "$D" --base-url https://example.invalid/dl/t
grep -q "unextractable body" "$T/out" && ok "refused as unextractable, not skipped" || bad "another reason: $(tail -1 "$T/out")"
printf 'echo "started; ls /opt/igneum\n' > "$T/lost-quote.sh"
expect_fail "a bash script with a lost quote" "$PUBLISH" add --kind run --target 1ccfe586 --script "$T/lost-quote.sh" --id test-lost-quote-sh --dest "$D" --base-url https://example.invalid/dl/t
printf '& wsl.exe -d Ubuntu-24.04 -u root -- bash /mnt/c/prove.sh\n# igneum-prove-host --mode shard --shard 0\n' > "$T/root-prover.ps1"
expect_fail "a root prover script without the socket cleanup" "$PUBLISH" add --kind run --target 1ccfe586 --script "$T/root-prover.ps1" --id test-root-socket --dest "$D" --base-url https://example.invalid/dl/t
grep -q "prover-socket" "$T/out" && ok "refused by the socket check" || bad "another reason: $(tail -1 "$T/out")"
printf '$jobs = Split-Path $env:IGNEUM_JOB_DIR\n$exe = Join-Path (Join-Path $jobs "fetch-kit-1") "worker.exe"\n& $exe --list\n' > "$T/kit-unchecked.ps1"
expect_fail "a run script that uses a fetched kit before a presence check" "$PUBLISH" add --kind run --target 1ccfe586 --script "$T/kit-unchecked.ps1" --id test-kit-unchecked --dest "$D" --base-url https://example.invalid/dl/t
grep -q "kit path used before a presence check" "$T/out" && ok "refused by the kit-path check" || bad "another reason: $(tail -1 "$T/out")"
cmp -s "$T/before.signed" "$D/igneum-jobs.signed.json" && ok "a refused publish leaves the envelope untouched" || bad "a refused publish changed the envelope"
echo "== the key"
EMB="$("$SIGNER" embedded | sed -n 1p)"
[ "$EMB" = "$(tr -d '[:space:]' < "$PUB")" ] && ok "the embedded key is the Mac's OTA public key" || bad "the embedded key is not $PUB"

View file

@ -9,7 +9,7 @@
#define ArtDir "..\..\brand\icons"
#endif
#ifndef AppVersion
#define AppVersion "0.3.10"
#define AppVersion "0.3.11"
#endif
#define AppName "Igneum Miner"
#define Publisher "Igneum"

View file

@ -1 +1 @@
21d4c73c6ce32fcbd68391968e85827339a511c0
89dfcb95be5ace1a2b7a4fb18d88aeb22023cab4

115
proto-cuda/dot4-probe.cu Normal file
View file

@ -0,0 +1,115 @@
// dot4-probe (CUDA): dp4a-class throughput on NVIDIA, standalone (no pack, no lottery kernel).
// Counter ASIC 2.0 layer 7 (docs/analysis/int8-matrix-family.md), 5 October 2026. PC job; not run on the Mac.
//
// Three dependent chains, same shape as the OpenCL --memprobe ALU chain (proto-opencl/host.c, probe_alu) and the
// Metal probe (proto-metal/dot4-probe.swift): 1,048,576 lanes x 4,096 steps, best of 3, event time.
// alu x = x * K + rotl(y, 7); y = (y ^ x) + s the card's integer baseline, 5 ops per step counted
// dot4i acc = __dp4a(x, y, acc) (PTX dp4a.s32.s32, sm_61+) one hardware dot4 per step per lane
// dot4e the scalar emulation of the same (4 sign-extended byte products summed, wrapping int32)
// The emulation and the intrinsic must agree bit for bit with the CPU reference (checked on two lanes per run).
//
// Build (Windows, CUDA Toolkit): nvcc -O2 -arch=sm_120 -o dot4-probe-cuda.exe dot4-probe.cu
// Build (Linux): nvcc -O2 -arch=sm_120 -o dot4-probe-cuda dot4-probe.cu
// Run: dot4-probe-cuda [--lanes N] [--steps N] [--reps N] [--device N]
#include <cuda_runtime.h>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <cstdint>
__host__ __device__ inline uint32_t pm_mix(uint32_t x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; }
__host__ __device__ inline uint32_t rotl32(uint32_t v, uint32_t n) { return (v << n) | (v >> (32u - n)); }
__host__ __device__ inline int32_t dot4_emul(uint32_t a, uint32_t b, int32_t acc) {
int32_t r = acc;
for (int i = 0; i < 4; ++i) {
int32_t ba = (int32_t)(int8_t)((a >> (8 * i)) & 0xffu);
int32_t bb = (int32_t)(int8_t)((b >> (8 * i)) & 0xffu);
r = (int32_t)((uint32_t)r + (uint32_t)(ba * bb));
}
return r;
}
__global__ void probe_alu(uint32_t steps, uint32_t seed, uint32_t* out) {
uint32_t g = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u;
for (uint32_t s = 0; s < steps; ++s) { x = x * 0x9E3779B1u + rotl32(y, 7u); y = (y ^ x) + s; }
out[g] = x ^ y;
}
__global__ void probe_dot4i(uint32_t steps, uint32_t seed, uint32_t* out) {
uint32_t g = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u;
int32_t acc = (int32_t)pm_mix(x);
for (uint32_t s = 0; s < steps; ++s) {
acc = __dp4a((int)x, (int)y, acc);
x = x * 0x9E3779B1u + (uint32_t)acc;
y = rotl32(y, 7u) ^ ((uint32_t)acc + s);
}
out[g] = (uint32_t)acc ^ x ^ y;
}
__global__ void probe_dot4e(uint32_t steps, uint32_t seed, uint32_t* out) {
uint32_t g = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u;
int32_t acc = (int32_t)pm_mix(x);
for (uint32_t s = 0; s < steps; ++s) {
acc = dot4_emul(x, y, acc);
x = x * 0x9E3779B1u + (uint32_t)acc;
y = rotl32(y, 7u) ^ ((uint32_t)acc + s);
}
out[g] = (uint32_t)acc ^ x ^ y;
}
static uint32_t lane_ref(const char* name, uint32_t g, uint32_t seed, uint32_t steps) {
uint32_t x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u;
if (strcmp(name, "alu") == 0) {
for (uint32_t s = 0; s < steps; ++s) { x = x * 0x9E3779B1u + rotl32(y, 7u); y = (y ^ x) + s; }
return x ^ y;
}
int32_t acc = (int32_t)pm_mix(x);
for (uint32_t s = 0; s < steps; ++s) {
acc = dot4_emul(x, y, acc);
x = x * 0x9E3779B1u + (uint32_t)acc;
y = rotl32(y, 7u) ^ ((uint32_t)acc + s);
}
return (uint32_t)acc ^ x ^ y;
}
#define CK(x) do { cudaError_t e = (x); if (e != cudaSuccess) { printf("CUDA error %s at %s:%d\n", cudaGetErrorString(e), __FILE__, __LINE__); return 1; } } while (0)
int main(int argc, char** argv) {
uint32_t lanes = 1u << 20, steps = 4096u; int reps = 3, device = 0;
for (int i = 1; i < argc; ++i) {
if (!strcmp(argv[i], "--lanes") && i + 1 < argc) lanes = (uint32_t)strtoul(argv[++i], 0, 10);
else if (!strcmp(argv[i], "--steps") && i + 1 < argc) steps = (uint32_t)strtoul(argv[++i], 0, 10);
else if (!strcmp(argv[i], "--reps") && i + 1 < argc) reps = atoi(argv[++i]);
else if (!strcmp(argv[i], "--device") && i + 1 < argc) device = atoi(argv[++i]);
else { printf("unknown argument %s\n", argv[i]); return 2; }
}
CK(cudaSetDevice(device));
cudaDeviceProp p; CK(cudaGetDeviceProperties(&p, device));
printf("dot4-probe (CUDA) on %s, sm_%d%d, %d SMs, %d MHz, lanes %u, steps %u, best of %d, event time\n", p.name, p.major, p.minor, p.multiProcessorCount, p.clockRate / 1000, lanes, steps, reps);
uint32_t* d_out; CK(cudaMalloc(&d_out, (size_t)lanes * 4));
uint32_t* h_out = (uint32_t*)malloc((size_t)lanes * 4);
cudaEvent_t e0, e1; CK(cudaEventCreate(&e0)); CK(cudaEventCreate(&e1));
printf("| kernel | lanes | steps | best ms | G steps/s (= G dot4/s for dot4 rows) | ns per dependent step | lanes 0 and last ok |\n|---|---|---|---|---|---|---|\n");
const char* names[3] = { "alu", "dot4i", "dot4e" };
for (int k = 0; k < 3; ++k) {
float best = 1e30f; int ok = 1;
for (int r = 0; r < reps; ++r) {
uint32_t seed = 0x2468aceu + (uint32_t)r * 0x9E3779B9u;
CK(cudaEventRecord(e0));
if (k == 0) probe_alu<<<lanes / 256, 256>>>(steps, seed, d_out);
else if (k == 1) probe_dot4i<<<lanes / 256, 256>>>(steps, seed, d_out);
else probe_dot4e<<<lanes / 256, 256>>>(steps, seed, d_out);
CK(cudaEventRecord(e1)); CK(cudaEventSynchronize(e1)); CK(cudaGetLastError());
float ms = 0; CK(cudaEventElapsedTime(&ms, e0, e1)); if (ms < best) best = ms;
CK(cudaMemcpy(h_out, d_out, (size_t)lanes * 4, cudaMemcpyDeviceToHost));
uint32_t gs[2] = { 0u, lanes - 1u };
for (int j = 0; j < 2; ++j) { uint32_t want = lane_ref(names[k], gs[j], seed, steps); if (h_out[gs[j]] != want) { ok = 0; printf("MISMATCH %s lane %u: gpu %08x cpu %08x\n", names[k], gs[j], h_out[gs[j]], want); } }
}
double sps = (double)lanes * (double)steps / (best / 1000.0);
printf("| %s | %u | %u | %.3f | %.2f | %.3f | %s |\n", names[k], lanes, steps, best, sps / 1e9, best * 1e6 / (double)steps, ok ? "yes" : "NO");
printf("RESULT DOT4 vendor=nvidia device=\"%s\" kernel=%s lanes=%u steps=%u best_ms=%.3f gsteps_per_s=%.2f ok=%d\n", p.name, names[k], lanes, steps, best, sps / 1e9, ok);
}
printf("dot4-probe: done\n");
return 0;
}

View file

@ -20,6 +20,9 @@
#define __forceinline__ inline
struct uint3 { unsigned x, y, z; };
// uint4 and make_uint4 (read-width experiment, 5 October 2026): the wide loads and the scratch slots are 16-byte vectors.
struct uint4 { unsigned x, y, z, w; };
static inline uint4 make_uint4(unsigned x, unsigned y, unsigned z, unsigned w) { uint4 v; v.x = x; v.y = y; v.z = z; v.w = w; return v; }
struct dim3 { unsigned x, y, z; dim3(unsigned x_ = 1, unsigned y_ = 1, unsigned z_ = 1) : x(x_), y(y_), z(z_) {} };
extern thread_local uint3 threadIdx;
extern thread_local uint3 blockIdx;

28
proto-cuda/emu/test-layout.sh Executable file
View file

@ -0,0 +1,28 @@
#!/usr/bin/env bash
# The host-side dataset derivation of the harnesses under a non-linear item-to-word layout (era layout,
# docs/plans/era-layout.md 1.2): proto-cuda/host.cu (and proto-opencl/host.c, the same function) derive dataset words
# on the host through the pack's own mh_word, which carries the layout. Until 5 October 2026 both derived
# mh_item(w >> 4)[w & 15] and the "64 random points vs host derivation" check failed on every interleaved pack while
# the Mac's samples and the vectors passed (the known-failed case, docs/bench-log.md, era layout entry). This script
# runs the CUDA CPU emulation on an interleaved era pack and a linear one and requires OVERALL: PASS on both; it is
# the test that fails on the old derivation. Usage: emu/test-layout.sh [pack dir ...]
set -euo pipefail
HERE="$(cd "$(dirname "$0")" && pwd)"
CUDA_DIR="$(cd "$HERE/.." && pwd)"
PACKS=("$@")
[ ${#PACKS[@]} -gt 0 ] || PACKS=("$CUDA_DIR/packs-ca2-era/era-1" "$CUDA_DIR/packs/igneum-devnet-v4-epoch0")
CXX="${CXX:-c++}"
for P in "${PACKS[@]}"; do
grep -q "mh_addr(" "$P/memhard.h" && kind=interleaved || kind=linear
BUILD="$HERE/build-layout-$(basename "$P")"
mkdir -p "$BUILD"
cp "$CUDA_DIR/host.cu" "$BUILD/host_emu.cpp"
sed -E 's/([A-Za-z_0-9]+)<<<([^,]+), ([^>]+)>>>\(/emu_launch(\1, \2, \3, /' "$P/kernel.cu" > "$BUILD/kernel_emu.cpp"
sed -E 's/([A-Za-z_0-9]+)<<<([^,]+), ([^>]+)>>>\(/emu_launch(\1, \2, \3, /' "$P/kernel_bound.cu" > "$BUILD/kernel_bound_emu.cpp"
"$CXX" -std=c++17 -O2 -w -I "$HERE" -I "$P" -o "$BUILD/igneum-emu" "$BUILD/host_emu.cpp" "$BUILD/kernel_emu.cpp" -DIGNEUM_BOUND "$BUILD/kernel_bound_emu.cpp" "$HERE/shim.cpp" -pthread
out="$("$BUILD/igneum-emu" --batch-log2 13 --batches 1 2>&1)"
echo "$out" | grep -E "dataset self-test|OVERALL" | sed "s|^|$(basename "$P") ($kind): |"
echo "$out" | grep -q "^OVERALL: PASS" || { echo "FAIL: $P ($kind layout) did not pass the emulated harness"; exit 1; }
echo "$out" | grep -q "64 random points vs host derivation PASS" || { echo "FAIL: $P ($kind layout): the host derivation disagrees with the pack's layout"; exit 1; }
done
echo "PASS: the host derivation follows the pack's layout on ${#PACKS[@]} pack(s)"

View file

@ -115,11 +115,11 @@ static bool setupCache() {
return gCachePass;
}
// dataset[w] derived on the host from the host cache, exactly as proto-metal's verifier does it.
// dataset[w] derived on the host from the host cache through the pack's own mh_word (memhard.h), which carries the
// pack's item-to-word layout (era layout, 5 October 2026: the harness's former w >> 4 / w & 15 failed the random
// points of every interleaved pack while the Mac samples and vectors passed).
static uint32_t host_ds_word(uint32_t w) {
uint32_t s[16];
mh_item(hCache.data(), w >> 4u, s);
return s[w & 15u];
return mh_word(hCache.data(), w);
}
#endif

View file

@ -103,11 +103,41 @@ int main(int argc, char** argv) {
{
char text[2000], sw[200], kw[200];
words_hex(bare, sw); words_hex(keyw, kw);
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, sw, kw);
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 2\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, sw, kw);
write_file(dir, "program.h", text);
write_file(dir, "seeds.txt", "epoch_seed_hex " EPOCH_34 "\nday_seed_hex " DAY_20731 "\n");
err[0] = 0;
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && pk.attempt == 0, "no attempt line reads as attempt 0 and the bare words load");
CHECK(strcmp(pk.programClass, "v2") == 0 && pk.eraHex[0] == 0, "a generator 2 pack without a class line is class v2 with no era");
}
// 4. Program classes (Counter ASIC 2.0, 5 October 2026, spec 01 section 1.4.5): a generator this worker does not
// run is refused; a generator 3 pack is class v3 and carries its era seed; a class line that contradicts the
// generator is refused; a pack with no generator line at all (generator 1, the retired lever generator) is refused.
{
char text[2400], sw[200], kw[200], why[256];
words_hex(bare, sw); words_hex(keyw, kw);
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 9\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, sw, kw);
write_file(dir, "program.h", text);
err[0] = 0;
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 0 && strstr(err, "generator 9 is not a generator version this worker runs") != NULL, "generator 9 is refused in plain words");
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 3\n#define IGNEUM_PROGRAM_CLASS \"v3\"\n#define IGNEUM_ERA_SEED_HEX \"%s\"\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, EPOCH_33, sw, kw);
write_file(dir, "program.h", text);
err[0] = 0;
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 1 && strcmp(pk.programClass, "v3") == 0 && strcmp(pk.eraHex, EPOCH_33) == 0, "a generator 3 pack loads as class v3 with its era seed");
CHECK(pf_pack_class_ok(pk.programClass, pk.eraHex, "v3", EPOCH_33, why, sizeof(why)) == 1, "the v3 pack matches a job naming class v3 and its era");
CHECK(pf_pack_class_ok(pk.programClass, pk.eraHex, "", "", why, sizeof(why)) == 1, "a job naming no class accepts the pack");
CHECK(pf_pack_class_ok(pk.programClass, pk.eraHex, "v2", "", why, sizeof(why)) == 0 && strstr(why, "program class mismatch") == why, "a job naming class v2 refuses the v3 pack");
CHECK(pf_pack_class_ok(pk.programClass, pk.eraHex, "v3", EPOCH_34, why, sizeof(why)) == 0 && strstr(why, "era seed mismatch") == why, "a job naming another era refuses the v3 pack");
CHECK(pf_pack_class_ok("v2", "", "v3", EPOCH_33, why, sizeof(why)) == 0, "a job naming class v3 refuses a v2 pack");
CHECK(pf_pack_class_ok("v2", "", "v2", EPOCH_33, why, sizeof(why)) == 1, "an era named on a v2 job is ignored");
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_GENERATOR 3\n#define IGNEUM_PROGRAM_CLASS \"v2\"\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, sw, kw);
write_file(dir, "program.h", text);
err[0] = 0;
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 0 && strstr(err, "does not match IGNEUM_GENERATOR 3") != NULL, "a class line that contradicts the generator is refused");
snprintf(text, sizeof(text), "#define IGNEUM_SEED_BYTES_HEX \"%s\"\n#define IGNEUM_DAY_BYTES_HEX \"%s\"\n#define IGNEUM_DATASET_LOG2 28\n#define IGNEUM_DATASET_MODE 1\n#define IGNEUM_SEEDW_INIT { %s }\n#define IGNEUM_KEY_INIT { %s }\n#define IGNEUM_CACHE_LOG2_WORDS 26\n#define IGNEUM_CACHE_SEGMENTS 4096u\n", EPOCH_34, DAY_20731, sw, kw);
write_file(dir, "program.h", text);
err[0] = 0;
CHECK(pf_load(dir, &pk, err, sizeof(err)) == 0 && strstr(err, "generator 1 is not") != NULL, "a pack with no generator line (generator 1) is refused");
}
printf("%s: %d failure(s)\n", argv[0], failures);

View file

@ -26,6 +26,21 @@ typedef struct {
uint32_t attempt; // IGNEUM_PROGRAM_ATTEMPT: seedw are the words of this attempt of the epoch seed (0 = bare seed)
uint32_t seedw[8], keyw[8];
char seedString[600];
// read-width experiment (5 October 2026): the load class (0 when absent), bytes per hash, variant 5's scratch
uint32_t loadsPerHash, bytesPerHash, scratchOps, persistent, scratchWordsPerLane;
char loadClass[64];
char programClass[8]; /* IGNEUM_PROGRAM_CLASS: "v2" or "v3" (Counter ASIC 2.0); absent = the generator's class */
char eraHex[65]; /* IGNEUM_ERA_SEED_HEX of a class v3 chain pack; empty otherwise */
// Counter ASIC 2.0 (5 October 2026): the mixer multiplier of the item derivation (IGNEUM_MIXER_MULT, 1 when absent:
// version 2; 4 under class v3). The emitted memhard.h / kernel.cl carry it in their text; this is for the log lines.
uint32_t mixerMult;
// hot-table experiment (5 October 2026, docs/plans/hot-table.md): hotMb 0 when the pack has no hot table; the
// table is filled on the device from the pack's igneum_hot_fill (never shipped), its self-test values from vectors.h
uint32_t hotMb, hotWords, hotSegments, hotSlots;
uint32_t hotKey[8];
int haveHot;
uint32_t hotHead[16], hotLast[16];
uint64_t hotFnv;
// seeds.txt (or program.h): the seeds as the worker protocol carries them
char epochHex[65];
char dayHex[PF_HEX_CAP];
@ -268,8 +283,36 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
if (!pf_define_u32(prog, "IGNEUM_CACHE_LOG2_WORDS", &pk->cacheLog2Words)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_LOG2_WORDS"); }
if (!pf_define_u32(prog, "IGNEUM_CACHE_SEGMENTS", &pk->cacheSegments)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_SEGMENTS"); }
if (!pf_define_u32(prog, "IGNEUM_GENERATOR", &pk->generator)) pk->generator = 1;
/* Spec 01 section 1.4.5: a pack whose generator version is not one this worker runs is refused. Generator 2 is
* program class v2 (the lottery hash of 4 October 2026), generator 3 is class v3 (Counter ASIC 2.0). */
if (pk->generator != 2 && pk->generator != 3) {
char m[200]; snprintf(m, sizeof(m), "program pack generator %u is not a generator version this worker runs (2 or 3)", (unsigned)pk->generator);
free(prog); return pf_fail(err, cap, m);
}
strcpy(pk->programClass, pk->generator == 3 ? "v3" : "v2");
{
char named[8] = {0};
if (pf_define_str(prog, "IGNEUM_PROGRAM_CLASS", named, sizeof(named)) && strcmp(named, pk->programClass) != 0) {
char m[200]; snprintf(m, sizeof(m), "program pack IGNEUM_PROGRAM_CLASS \"%.7s\" does not match IGNEUM_GENERATOR %u", named, (unsigned)pk->generator);
free(prog); return pf_fail(err, cap, m);
}
}
pk->eraHex[0] = 0; pf_define_str(prog, "IGNEUM_ERA_SEED_HEX", pk->eraHex, sizeof(pk->eraHex));
if (!pf_define_u32(prog, "IGNEUM_PROGRAM_ATTEMPT", &pk->attempt)) pk->attempt = 0;
if (pf_define_words(prog, "IGNEUM_SEEDW_INIT", pk->seedw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_SEEDW_INIT with 8 words"); }
pk->loadsPerHash = 128; pf_define_u32(prog, "IGNEUM_LOADS_PER_HASH", &pk->loadsPerHash);
pk->bytesPerHash = pk->loadsPerHash * 4u; pf_define_u32(prog, "IGNEUM_BYTES_PER_HASH", &pk->bytesPerHash);
pk->scratchOps = 0; pf_define_u32(prog, "IGNEUM_SCRATCH_OPS", &pk->scratchOps);
pk->persistent = 0; pf_define_u32(prog, "IGNEUM_PERSISTENT_WARPS", &pk->persistent);
pk->scratchWordsPerLane = 8192; pf_define_u32(prog, "IGNEUM_SCRATCH_WORDS_PER_LANE", &pk->scratchWordsPerLane);
strcpy(pk->loadClass, "v2"); pf_define_str(prog, "IGNEUM_LOAD_CLASS", pk->loadClass, sizeof(pk->loadClass));
pk->mixerMult = 1; pf_define_u32(prog, "IGNEUM_MIXER_MULT", &pk->mixerMult);
pk->hotMb = 0; pf_define_u32(prog, "IGNEUM_HOT_MB", &pk->hotMb);
if (pk->hotMb) {
if (!pf_define_u32(prog, "IGNEUM_HOT_WORDS", &pk->hotWords) || !pf_define_u32(prog, "IGNEUM_HOT_SEGMENTS", &pk->hotSegments) ||
!pf_define_u32(prog, "IGNEUM_HOT_SLOTS", &pk->hotSlots) || pf_define_words(prog, "IGNEUM_HOT_KEY_INIT", pk->hotKey, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has IGNEUM_HOT_MB but not IGNEUM_HOT_WORDS, IGNEUM_HOT_SEGMENTS, IGNEUM_HOT_SLOTS and IGNEUM_HOT_KEY_INIT"); }
if (pk->hotWords != pk->hotMb * 262144u || pk->hotSegments != pk->hotMb * 256u || pk->hotMb > 4096u) { free(prog); return pf_fail(err, cap, "program.h hot table sizes disagree (words must be MiB x 2^18, segments MiB x 256)"); }
}
if (pf_define_words(prog, "IGNEUM_KEY_INIT", pk->keyw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_KEY_INIT with 8 words"); }
if (!pf_define_str(prog, "IGNEUM_SEED_STRING", pk->seedString, sizeof(pk->seedString))) strncpy(pk->seedString, "(no IGNEUM_SEED_STRING)", sizeof(pk->seedString) - 1);
pf_define_str(prog, "IGNEUM_SEED_BYTES_HEX", ehex, sizeof(ehex));
@ -290,15 +333,19 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
strcpy(ehex, e2); strcpy(dhex, d2);
}
if (!ehex[0] || !dhex[0]) return pf_fail(err, cap, "no seeds: neither seeds.txt nor IGNEUM_SEED_BYTES_HEX / IGNEUM_DAY_BYTES_HEX in program.h (a pack from igneum-pow export --seed <name> has no byte seeds)");
if (strlen(ehex) != 64) return pf_fail(err, cap, "epoch seed is not 64 hex characters");
// The chain's epoch seed is 32 bytes (64 hex characters). A pack exported from a seed STRING (igneum-pow export
// --seed <name>, the read-width experiment's packs of 5 October 2026) carries the string's bytes instead; the
// seed-word re-derivation below checks either form, so any even-length hex seed is accepted here. The serve
// protocol still carries 64-hex seeds; a string-seed pack can only be benched (--bench, --bench-pack, --check).
if (strlen(ehex) < 2 || strlen(ehex) % 2 != 0) return pf_fail(err, cap, "epoch seed is not an even-length hex string");
strcpy(pk->epochHex, ehex); strcpy(pk->dayHex, dhex);
// The seed words derived from the bytes AND the attempt must be the pack's own words: otherwise the pack and
// its seeds disagree (a half rewritten directory, or an exporter on another rule)
{
uint32_t w[8];
char m[400];
if (!pf_unhex(ehex, bytes, 32, &blen) || blen != 32) return pf_fail(err, cap, "epoch seed hex is malformed");
pf_program_words(bytes, 32, pk->attempt, w);
if (!pf_unhex(ehex, bytes, sizeof(bytes), &blen) || blen == 0) return pf_fail(err, cap, "epoch seed hex is malformed");
pf_program_words(bytes, blen, pk->attempt, w);
if (memcmp(w, pk->seedw, 32) != 0) {
snprintf(m, sizeof(m), "program pack and its seeds disagree: IGNEUM_SEEDW_INIT is not attempt %u of the epoch seed %.16s (attempt %u gives %08x %08x ..., the pack has %08x %08x ...); run igneum-miner export-pack again",
(unsigned)pk->attempt, ehex, (unsigned)pk->attempt, w[0], w[1], pk->seedw[0], pk->seedw[1]);
@ -330,6 +377,9 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
pf_symbol_numbers(vec, "IGNEUM_CACHE_FNV64", &pk->cacheFnv, 1) == 1 && pk->vecWarps > 0) {
uint32_t ns = 0;
pk->haveVectors = 1;
if (pk->hotMb && pf_symbol_u32s(vec, "IGNEUM_HOT_HEAD", pk->hotHead, 16) == 16 &&
pf_symbol_u32s(vec, "IGNEUM_HOT_LAST", pk->hotLast, 16) == 16 &&
pf_symbol_numbers(vec, "IGNEUM_HOT_FNV64", &pk->hotFnv, 1) == 1) pk->haveHot = 1;
if (pf_define_u32(vec, "IGNEUM_DS_SAMPLES", &ns) && ns > 0 && ns <= PF_MAX_SAMPLES) {
int a = pf_symbol_u32s(vec, "IGNEUM_DS_SAMPLE_INDEX", pk->sampleIdx, (int)ns);
int b = pf_symbol_u32s(vec, "IGNEUM_DS_SAMPLE_VALUE", pk->sampleVal, (int)ns);
@ -343,23 +393,32 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
// The self-test verdict from values the host read back from the device. `vec` holds vecWarps x 32 outputs of the
// bound kernel run with the pack's own seed words as init words (that is igneum_hash of kernel.cu). Writes one line.
// A hot-table pack (pk->hotMb) also hands the hot table's head, last line and FNV-1a 64 (NULL and 0 otherwise); a
// hot pack whose vectors.h carries no hot values is not checked on the table (the vectors cover it) and says so.
static int pf_selftest(const PfPack* pk, const uint32_t* cacheHead, const uint32_t* cacheLast, uint64_t cacheFnv,
const uint32_t* dsHead, uint32_t dsLast, const uint32_t* sampleVals, const uint64_t* vec,
const uint32_t* hotHead, const uint32_t* hotLast, uint64_t hotFnv,
char* out, size_t cap) {
int okCH = memcmp(cacheHead, pk->cacheHead, 64) == 0, okCL = memcmp(cacheLast, pk->cacheLast, 64) == 0;
int okFnv = (cacheFnv == pk->cacheFnv);
int okDH = memcmp(dsHead, pk->dsHead, 64) == 0, okDL = (dsLast == pk->dsLast);
int hotChecked = (pk->hotMb && pk->haveHot && hotHead && hotLast);
int okHot = !hotChecked || (memcmp(hotHead, pk->hotHead, 64) == 0 && memcmp(hotLast, pk->hotLast, 64) == 0 && hotFnv == pk->hotFnv);
int badS = 0, badV = 0, i, w, l, firstBadWarp = -1, firstBadLane = -1;
for (i = 0; i < pk->nSamples; ++i) if (sampleVals[i] != pk->sampleVal[i]) ++badS;
for (w = 0; w < pk->vecWarps; ++w) for (l = 0; l < 32; ++l) if (vec[w * 32 + l] != pk->vecOut[w][l]) { if (firstBadWarp < 0) { firstBadWarp = w; firstBadLane = l; } ++badV; }
if (okCH && okCL && okFnv && okDH && okDL && badS == 0 && badV == 0) {
snprintf(out, cap, "self-test PASS (cache head, last line and FNV-1a 64 %016llx; dataset head, word [%u] and %d samples; %d of %d vector lanes)",
(unsigned long long)cacheFnv, pk->dsLastIndex, pk->nSamples, pk->vecWarps * 32, pk->vecWarps * 32);
if (okCH && okCL && okFnv && okDH && okDL && okHot && badS == 0 && badV == 0) {
if (pk->hotMb)
snprintf(out, cap, "self-test PASS (cache head, last line and FNV-1a 64 %016llx; dataset head, word [%u] and %d samples; hot table %u MiB %s; %d of %d vector lanes)",
(unsigned long long)cacheFnv, pk->dsLastIndex, pk->nSamples, pk->hotMb, hotChecked ? "head, last line and FNV-1a 64 ok" : "not in vectors.h (the vectors cover it)", pk->vecWarps * 32, pk->vecWarps * 32);
else
snprintf(out, cap, "self-test PASS (cache head, last line and FNV-1a 64 %016llx; dataset head, word [%u] and %d samples; %d of %d vector lanes)",
(unsigned long long)cacheFnv, pk->dsLastIndex, pk->nSamples, pk->vecWarps * 32, pk->vecWarps * 32);
return 1;
}
snprintf(out, cap, "self-test FAIL (cache head %s, cache last %s, cache FNV %016llx vs pack %016llx %s, dataset head %s, dataset last %s, samples %d bad of %d, vector lanes %d bad of %d%s)",
snprintf(out, cap, "self-test FAIL (cache head %s, cache last %s, cache FNV %016llx vs pack %016llx %s, dataset head %s, dataset last %s, samples %d bad of %d, hot table %s, vector lanes %d bad of %d%s)",
okCH ? "ok" : "BAD", okCL ? "ok" : "BAD", (unsigned long long)cacheFnv, (unsigned long long)pk->cacheFnv, okFnv ? "ok" : "BAD",
okDH ? "ok" : "BAD", okDL ? "ok" : "BAD", badS, pk->nSamples, badV, pk->vecWarps * 32,
okDH ? "ok" : "BAD", okDL ? "ok" : "BAD", badS, pk->nSamples, hotChecked ? (okHot ? "ok" : "BAD") : "none", badV, pk->vecWarps * 32,
firstBadWarp >= 0 ? " (first bad lane in the warp at base nonce" : "");
if (firstBadWarp >= 0) {
size_t n = strlen(out);
@ -369,4 +428,29 @@ static int pf_selftest(const PfPack* pk, const uint32_t* cacheHead, const uint32
return 0;
}
/* Counter ASIC 2.0 (5 October 2026): a job or prepare line may end with `class=<v2|v3>` and `era=<hex>` tokens (sent
* only when the chain is on class v3, so every v2 line is the line of before). A pack matches the line when its class
* is the named class and, when an era is named, its era seed is that era. Empty wanted strings accept any pack.
* Returns 1 on a match, else 0 with the reason in `why`. */
static int pf_pack_class_ok(const char* packClass, const char* packEra, const char* wantClass, const char* wantEra, char* why, size_t cap) {
if (wantClass && wantClass[0] && strcmp(wantClass, packClass) != 0) {
snprintf(why, cap, "program class mismatch: this pack is class %s, the job names class %s (export the pack again)", packClass, wantClass);
return 0;
}
if (wantEra && wantEra[0] && strcmp(packClass, "v3") == 0) {
size_t i; int eq = strlen(packEra) == strlen(wantEra);
for (i = 0; eq && packEra[i]; ++i) if (tolower((unsigned char)packEra[i]) != tolower((unsigned char)wantEra[i])) eq = 0;
if (!eq) { snprintf(why, cap, "era seed mismatch: this pack was drawn under era %.16s, the job names era %.16s (export the pack again)", packEra[0] ? packEra : "(none)", wantEra); return 0; }
}
return 1;
}
/* Reads a `class=` or `era=` token into `cls` / `era` (small fixed buffers). Returns 1 when the token was one of them. */
static int pf_class_token(const char* tok, char* cls, size_t clsCap, char* era, size_t eraCap) {
if (strncmp(tok, "class=", 6) == 0) { strncpy(cls, tok + 6, clsCap - 1); cls[clsCap - 1] = 0; return 1; }
if (strncmp(tok, "era=", 4) == 0) { strncpy(era, tok + 4, eraCap - 1); era[eraCap - 1] = 0; return 1; }
return 0;
}
#endif

View file

@ -241,6 +241,10 @@ static bool loadNvrtc(Rtc& r, std::string& err, std::string& libName) {
struct Ctx {
Drv drv;
Rtc rtc;
// read-width experiment, variant 5 (5 October 2026): persistent warps and their scratch; --warps caps the launch
int warps = 0; // 0 = the resident capacity from the occupancy query, rounded down to a power of two
uint32_t salt = 1; // the running per-unit tag salt (+= units per launch)
int batches = 5; // --bench: timed dispatches
CUdevice dev = 0;
CUcontext ctx = nullptr;
std::string name;
@ -421,12 +425,45 @@ struct Pair {
std::string variant = "base"; // the bound kernel in service: a variant name (see allVariants)
std::string raceLine; // the race's one-line report, emitted by the main thread with "prepared"
double raceMs = 0;
// read-width experiment (5 October 2026): the pack's load class and, for variant 5, the persistent-warp scratch
std::string loadClass = "v2";
std::string programClass = "v2", eraHex; // Counter ASIC 2.0: the pack's class and era seed (packfile.h)
uint32_t loadsPerHash = 128, bytesPerHash = 512, scratchOps = 0;
bool persistent = false;
CUdeviceptr scratch = 0;
int warps = 0; // persistent warps launched (the arena holds this many)
int residentWarps = 0; // the occupancy query's capacity: blocks/SM x warps/block x SMs
size_t scratchBytes = 0;
// hot-table experiment (5 October 2026, docs/plans/hot-table.md): the epoch's hot table, filled on the device by the
// pack's igneum_hot_fill, the argument after the init words
uint32_t hotMb = 0, hotWords = 0, hotSegments = 0, hotSlots = 0;
CUfunction fHotFill = nullptr;
CUdeviceptr hot = 0;
double hotMs = 0;
};
static bool pairIs(const Pair* p, const std::string& epochHex, const std::string& dayHex) {
return p && hexEq(p->epochHex, epochHex) && hexEq(p->dayHex, dayHex);
}
// Counter ASIC 2.0: a job that names a class (and an era) belongs to a pair of that class (and era) only, so a pack
// of the old class for the same seeds is not this job's pair and the prepared pack of the right class wins.
static bool pairIsClass(const Pair* p, const std::string& epochHex, const std::string& dayHex, const std::string& cls, const std::string& era) {
char why[256];
return pairIs(p, epochHex, dayHex) && pf_pack_class_ok(p->programClass.c_str(), p->eraHex.c_str(), cls.c_str(), era.c_str(), why, sizeof(why));
}
// The trailing `class=` and `era=` tokens of a job or prepare line (absent on every class v2 line), removed from `f`.
static void takeClassTokens(std::vector<std::string>& f, std::string& cls, std::string& era) {
while (!f.empty()) {
char c[8] = {0}, e[65] = {0};
if (!pf_class_token(f.back().c_str(), c, sizeof(c), e, sizeof(e))) break;
if (c[0]) cls = c;
if (e[0]) era = e;
f.pop_back();
}
}
// The job loop and a race take turns on the card: a variant is timed with no job running (exclusive numbers), and
// mining resumes between variants. Held per chunk by the job loop, per variant by the race.
static std::mutex gpuMutex;
@ -435,6 +472,8 @@ static void releasePair(Ctx& c, Pair* p) {
if (!p) return;
if (p->ds) c.drv.memFree(p->ds);
if (p->cache) c.drv.memFree(p->cache);
if (p->scratch) c.drv.memFree(p->scratch);
if (p->hot) c.drv.memFree(p->hot);
if (p->modBound) c.drv.moduleUnload(p->modBound);
if (p->modKernel) c.drv.moduleUnload(p->modKernel);
delete p;
@ -446,7 +485,20 @@ struct IgneumInitWordsArg { uint32_t w[8]; };
static bool launchHash(Ctx& c, Pair* p, CUdeviceptr out, uint32_t baseNonce, const uint32_t iw[8], uint32_t nonces, uint32_t block, CUstream s, std::string& err) {
uint32_t mask = p->words - 1u;
IgneumInitWordsArg a; std::memcpy(a.w, iw, 32);
void* args[5] = { &p->ds, &out, &baseNonce, &mask, &a };
if (p->persistent) {
// Variant 5: N persistent warps over nonces / 32 units; the arena was sized for p->warps warps in buildPair.
uint32_t units = nonces / 32u, warps = (uint32_t)p->warps;
if (warps > units) warps = units;
while (warps > 1u && units % warps != 0u) warps >>= 1;
if (block != 32u) { err = "a variant-5 pack runs one warp per block (--block-warps 1)"; return false; }
uint32_t salt = c.salt; c.salt += units;
// the hot table (when the pack has one) sits between the init words and the scratch triple
void* args[9] = { &p->ds, &out, &baseNonce, &mask, &a, &p->scratch, &units, &salt, nullptr };
if (p->hot) { args[5] = &p->hot; args[6] = &p->scratch; args[7] = &units; args[8] = &salt; }
DRV_CHECK(c, c.drv.launchKernel(p->fHashBound, warps, 1, 1, 32, 1, 1, 0, s, args, nullptr), "cuLaunchKernel igneum_hash_bound (persistent)");
return true;
}
void* args[6] = { &p->ds, &out, &baseNonce, &mask, &a, &p->hot };
DRV_CHECK(c, c.drv.launchKernel(p->fHashBound, nonces / block, 1, 1, block, 1, 1, 0, s, args, nullptr), "cuLaunchKernel igneum_hash_bound");
return true;
}
@ -769,7 +821,9 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
p->datasetLog2 = pk.datasetLog2; p->words = 1u << pk.datasetLog2; p->cacheWords = 1u << pk.cacheLog2Words; p->cacheSegments = pk.cacheSegments;
// Compile
Compiled ck, cb;
if (!rtcCompile(c, kernelDev, "kernel.cu", programH, memhardH, { "igneum_cache_fill", "igneum_build" }, ck, err)) { releasePair(c, p); return nullptr; }
std::vector<std::string> kernelNames = { "igneum_cache_fill", "igneum_build" };
if (pk.hotMb) kernelNames.push_back("igneum_hot_fill");
if (!rtcCompile(c, kernelDev, "kernel.cu", programH, memhardH, kernelNames, ck, err)) { releasePair(c, p); return nullptr; }
if (!rtcCompile(c, boundDev, "kernel_bound.cu", programH, memhardH, { "igneum_hash_bound" }, cb, err)) { releasePair(c, p); return nullptr; }
p->compileMs = ck.ms + cb.ms;
// Load
@ -781,19 +835,45 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
if (c.drv.moduleGetFunction(&p->fCacheFill, p->modKernel, ck.lowered[0].c_str()) != CUDA_SUCCESS) { err = "igneum_cache_fill (" + ck.lowered[0] + ") not in the module"; releasePair(c, p); return nullptr; }
if (c.drv.moduleGetFunction(&p->fBuild, p->modKernel, ck.lowered[1].c_str()) != CUDA_SUCCESS) { err = "igneum_build (" + ck.lowered[1] + ") not in the module"; releasePair(c, p); return nullptr; }
if (c.drv.moduleGetFunction(&p->fHashBound, p->modBound, cb.lowered[0].c_str()) != CUDA_SUCCESS) { err = "igneum_hash_bound (" + cb.lowered[0] + ") not in the module"; releasePair(c, p); return nullptr; }
if (pk.hotMb && c.drv.moduleGetFunction(&p->fHotFill, p->modKernel, ck.lowered[2].c_str()) != CUDA_SUCCESS) { err = "igneum_hot_fill (" + ck.lowered[2] + ") not in the module"; releasePair(c, p); return nullptr; }
c.drv.funcGetAttribute(&p->regs, CU_FUNC_ATTRIBUTE_NUM_REGS, p->fHashBound);
c.drv.occupancy(&p->blocksPerSM, p->fHashBound, 32 * c.blockWarps, 0);
}
p->loadClass = pk.loadClass; p->loadsPerHash = pk.loadsPerHash; p->bytesPerHash = pk.bytesPerHash; p->scratchOps = pk.scratchOps;
p->programClass = pk.programClass; p->eraHex = pk.eraHex;
p->persistent = pk.persistent != 0;
p->hotMb = pk.hotMb; p->hotWords = pk.hotWords; p->hotSegments = pk.hotSegments; p->hotSlots = pk.hotSlots;
p->residentWarps = p->blocksPerSM * c.blockWarps * c.sms;
size_t scratchBytes = 0, hotBytes = (size_t)pk.hotWords * 4u;
if (p->persistent) {
// Variant 5: one arena per launched warp. The launch is the resident capacity (the occupancy query), rounded
// down to a power of two so it divides every batch, or --warps; the allocation cannot change the occupancy
// (registers and shared memory decide it), and the number is re-queried after the allocation below to show it.
if (c.blockWarps != 1) { err = "a variant-5 pack runs one warp per block: use --block-warps 1"; releasePair(c, p); return nullptr; }
int w = c.warps > 0 ? c.warps : p->residentWarps;
int pw = 1; while (pw * 2 <= w) pw *= 2;
p->warps = pw;
scratchBytes = (size_t)p->warps * 32u * (size_t)pk.scratchWordsPerLane * 4u;
}
// Cache
double t0 = wallMs();
size_t cacheBytes = (size_t)p->cacheWords * 4u, dsBytes = (size_t)p->words * 4u;
{
size_t freeB = 0, totalB = 0;
if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS && freeB < cacheBytes + dsBytes + (64u << 20)) {
err = fmt("%llu MiB free on the device, this pack needs %llu MiB (cache %llu + dataset %llu)", (unsigned long long)(freeB >> 20), (unsigned long long)((cacheBytes + dsBytes) >> 20), (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20));
if (c.drv.memGetInfo(&freeB, &totalB) == CUDA_SUCCESS && freeB < cacheBytes + dsBytes + scratchBytes + hotBytes + (64u << 20)) {
err = fmt("%llu MiB free on the device, this pack needs %llu MiB (cache %llu + dataset %llu + scratch %llu + hot %llu)", (unsigned long long)(freeB >> 20), (unsigned long long)((cacheBytes + dsBytes + scratchBytes + hotBytes) >> 20), (unsigned long long)(cacheBytes >> 20), (unsigned long long)(dsBytes >> 20), (unsigned long long)(scratchBytes >> 20), (unsigned long long)(hotBytes >> 20));
releasePair(c, p); return nullptr;
}
}
if (p->persistent) {
CUresult r = c.drv.memAlloc(&p->scratch, scratchBytes);
if (r != CUDA_SUCCESS) { err = "cuMemAlloc scratch: " + c.err(r); p->scratch = 0; releasePair(c, p); return nullptr; }
p->scratchBytes = scratchBytes;
int after = 0;
c.drv.occupancy(&after, p->fHashBound, 32 * c.blockWarps, 0);
info(fmt("variant 5: %d persistent warps (resident capacity %d = %d blocks/SM x %d warps/block x %d SMs; occupancy query after the allocation %d blocks/SM), scratch %llu MiB (%u KiB per warp)",
p->warps, p->residentWarps, p->blocksPerSM, c.blockWarps, c.sms, after, (unsigned long long)(scratchBytes >> 20), pk.scratchWordsPerLane * 4u * 32u / 1024u));
}
{
CUresult r = c.drv.memAlloc(&p->cache, cacheBytes);
if (r != CUDA_SUCCESS) { err = "cuMemAlloc cache: " + c.err(r); p->cache = 0; releasePair(c, p); return nullptr; }
@ -816,6 +896,18 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
if (r != CUDA_SUCCESS) { err = "dataset build: " + c.err(r); releasePair(c, p); return nullptr; }
}
p->dsMs = wallMs() - t0;
// Hot table (hot-table experiment): filled from the epoch seed by the pack's own kernel, never shipped
if (pk.hotMb) {
t0 = wallMs();
CUresult r = c.drv.memAlloc(&p->hot, hotBytes);
if (r != CUDA_SUCCESS) { err = "cuMemAlloc hot table: " + c.err(r); p->hot = 0; releasePair(c, p); return nullptr; }
uint32_t nSeg = pk.hotSegments, block = 256u, grid = (nSeg + block - 1u) / block;
void* args[2] = { &p->hot, &nSeg };
r = c.drv.launchKernel(p->fHotFill, grid, 1, 1, block, 1, 1, 0, s, args, nullptr);
if (r == CUDA_SUCCESS) r = c.drv.streamSynchronize(s);
if (r != CUDA_SUCCESS) { err = "hot table fill: " + c.err(r); releasePair(c, p); return nullptr; }
p->hotMs = wallMs() - t0;
}
// Self-test against vectors.h
t0 = wallMs();
if (!pk.haveVectors) {
@ -844,8 +936,17 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
if ((r = c.drv.streamSynchronize(s)) != CUDA_SUCCESS || (r = c.drv.memcpyDtoH(&vec[(size_t)w * 32u], out, 32u * 8u)) != CUDA_SUCCESS) { err = "vector warp: " + c.err(r); c.drv.memFree(out); releasePair(c, p); return nullptr; }
}
c.drv.memFree(out);
uint32_t hotHead[16] = {0}, hotLast[16] = {0};
uint64_t hotFnv = 0;
if (p->hot) {
std::vector<uint32_t> hw(p->hotWords);
if ((r = c.drv.memcpyDtoH(hw.data(), p->hot, hotBytes)) != CUDA_SUCCESS) { err = "cuMemcpyDtoH hot table: " + c.err(r); releasePair(c, p); return nullptr; }
std::memcpy(hotHead, hw.data(), 64);
std::memcpy(hotLast, hw.data() + p->hotWords - 16u, 64);
hotFnv = pf_fnv1a64(hw.data(), hotBytes);
}
char line[1024];
p->checkPass = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples.data(), vec.data(), line, sizeof(line)) != 0;
p->checkPass = pf_selftest(&pk, cacheHead, cacheLast, fnv, dsHead, dsLast, samples.data(), vec.data(), p->hot ? hotHead : nullptr, p->hot ? hotLast : nullptr, hotFnv, line, sizeof(line)) != 0;
p->checked = true;
p->check = line;
}
@ -857,7 +958,7 @@ static Pair* buildPair(Ctx& c, const std::string& dir, CUstream s, std::string&
}
static std::string pairSummary(const Pair* p) {
return fmt("nvrtc %.0f cache %.0f dataset %.0f check %.0f race %.0f ms variant %s; %s", p->compileMs, p->cacheMs, p->dsMs, p->checkMs, p->raceMs, p->variant.c_str(), p->check.c_str());
return fmt("nvrtc %.0f cache %.0f dataset %.0f hot %.0f check %.0f race %.0f ms variant %s; %s", p->compileMs, p->cacheMs, p->dsMs, p->hotMs, p->checkMs, p->raceMs, p->variant.c_str(), p->check.c_str());
}
// ---------------------------------------------------------------------------------------------
@ -865,6 +966,7 @@ static std::string pairSummary(const Pair* p) {
struct PrepareTask {
std::string epochHex, dayHex, dir, error;
std::string wantClass, wantEra; // the class and era the prepare line named (empty: any)
std::atomic<bool> done{false};
Pair* result = nullptr;
double t0 = 0;
@ -882,6 +984,11 @@ static void prepareRun(Ctx* c, PrepareTask* t) {
err = "the pack in " + t->dir + " is for epoch " + p->epochHex.substr(0, 16) + " day " + p->dayHex + ", not the prepared seeds";
releasePair(*c, p); p = nullptr;
}
char why[256];
if (p && !pf_pack_class_ok(p->programClass.c_str(), p->eraHex.c_str(), t->wantClass.c_str(), t->wantEra.c_str(), why, sizeof(why))) {
err = "pack " + t->dir + ": " + why;
releasePair(*c, p); p = nullptr;
}
t->result = p;
t->error = err;
t->done = true;
@ -892,6 +999,8 @@ static void prepareRun(Ctx* c, PrepareTask* t) {
struct Options {
bool serve = false, check = false, raceOnly = false;
bool bench = false, memprobe = false; // read-width experiment (5 October 2026)
int batches = 5, warps = 0, probeMib = 0;
int device = 0, batchLog2 = 22, blockWarps = 1;
std::string pack, arch = "auto";
std::string race = "on", pinned, tuningPath;
@ -906,6 +1015,12 @@ static void usage() {
" --batch-log2 B nonces per dispatch = 2^B (default 22)\n"
" --block-warps W warps per thread block (default 1)\n"
" --arch sm_XY|compute_XY|auto NVRTC target (default auto: the device's architecture)\n"
" --bench --pack <dir> read-width experiment: build and self-test the pack, time --batches dispatches of 2^B nonces,\n"
" print the 2^B fingerprint at base nonce 0 (one RESULT line); a variant-5 pack runs --warps persistent warps\n"
" --memprobe [--probe-mib N] no pack: dependent random 4, 16 and 64-byte reads, independent reads, a coalesced stream and an\n"
" integer chain at 4, 64 and 1024 MiB (the same table as igneum-worker-opencl --memprobe)\n"
" --batches N --bench: timed dispatches (default 5)\n"
" --warps N --bench on a variant-5 pack: persistent warps (default: the occupancy capacity, rounded down to a power of two)\n"
" --race --pack <dir> the variant race alone (3 rounds): one line per variant, the race line, exit 0 or 1\n"
" --race on|off|a,b,c in --serve: race every variant (default), none, or these names\n"
" --race-bench-ms N timed window per variant (default 2000)\n"
@ -922,6 +1037,11 @@ static Options parseArgs(int argc, char** argv) {
auto next = [&]() -> std::string { if (i + 1 >= argc) { usage(); std::exit(2); } return argv[++i]; };
if (a == "--serve") o.serve = true;
else if (a == "--check") o.check = true;
else if (a == "--bench") o.bench = true;
else if (a == "--memprobe") o.memprobe = true;
else if (a == "--batches") o.batches = std::atoi(next().c_str());
else if (a == "--warps") o.warps = std::atoi(next().c_str());
else if (a == "--probe-mib") o.probeMib = std::atoi(next().c_str());
else if (a == "--race" && (i + 1 >= argc || std::string(argv[i + 1]).rfind("--", 0) == 0)) o.raceOnly = true;
else if (a == "--race") o.race = next();
else if (a == "--race-bench-ms") o.raceBenchMs = std::atoi(next().c_str());
@ -940,12 +1060,12 @@ static Options parseArgs(int argc, char** argv) {
}
if (o.batchLog2 < 10 || o.batchLog2 > 28) { std::printf("--batch-log2 must be between 10 and 28\n"); std::exit(2); }
if (o.blockWarps < 1 || o.blockWarps > 32) { std::printf("--block-warps must be between 1 and 32\n"); std::exit(2); }
if (!o.serve && !o.check && !o.raceOnly) { usage(); std::exit(2); }
if (!o.serve && !o.check && !o.raceOnly && !o.bench && !o.memprobe) { usage(); std::exit(2); }
if (o.raceBenchMs < 200 || o.raceBenchMs > 20000) { std::printf("--race-bench-ms must be between 200 and 20000\n"); std::exit(2); }
if (o.raceBudgetS < 5 || o.raceBudgetS > 540) { std::printf("--race-budget-s must be between 5 and 540 (the prepare lead is 600 DAA)\n"); std::exit(2); }
if (o.raceRounds == 0) o.raceRounds = o.raceOnly ? 3 : 1;
if (o.tuningPath.empty()) if (const char* t = std::getenv("IGNEUM_TUNING_FILE")) o.tuningPath = t;
if (o.pack.empty()) { std::printf("--pack <dir> is required (igneum-miner export-pack <node> <dir> writes one)\n"); std::exit(2); }
if (o.pack.empty() && !o.memprobe) { std::printf("--pack <dir> is required (igneum-miner export-pack <node> <dir> writes one)\n"); std::exit(2); }
while (o.pack.size() > 1 && (o.pack.back() == '/' || o.pack.back() == '\\')) o.pack.pop_back();
return o;
}
@ -1023,6 +1143,8 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
}
delete task; task = nullptr;
}
std::string wantClass, wantEra;
takeClassTokens(f, wantClass, wantEra);
if (f[0] == "prepare") {
if (f.size() < 4) { emit(fmt("prepare-failed %s %s a pack directory is needed as the third field (igneum-miner --prepare-packs <dir>)", f.size() > 1 ? f[1].c_str() : "0", f.size() > 2 ? f[2].c_str() : "0")); continue; }
if (f[1].size() != 64) { emit(fmt("prepare-failed %s %s bad field (epoch_seed 64 hex, day_seed hex)", f[1].c_str(), f[2].c_str())); continue; }
@ -1034,6 +1156,7 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
prepareRoot = parentDir(dir);
task = new PrepareTask();
task->epochHex = f[1]; task->dayHex = f[2]; task->dir = dir; task->t0 = wallMs();
task->wantClass = wantClass; task->wantEra = wantEra;
task->thread = std::thread(prepareRun, &c, task);
info(fmt("prepare started for epoch %.16s day %s from %s (NVRTC %s in the background)", f[1].c_str(), f[2].c_str(), dir.c_str(), c.archOpt.c_str()));
continue;
@ -1055,7 +1178,7 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
pf_seed_words_from_bytes(daySeed, dl, kw);
double t0 = wallMs();
bool switched = false;
if (!pairIs(cur, f[6], f[7]) && !task && !pairIs(prepared, f[6], f[7])) {
if (!pairIsClass(cur, f[6], f[7], wantClass, wantEra) && !task && !pairIsClass(prepared, f[6], f[7], wantClass, wantEra)) {
// Self-heal: a job on seeds this worker has no pair for and no prepare in flight (a prepare failed, or
// the miner never sent one). The miner writes a pack per pair under its --prepare-packs root; find it by
// seeds.txt and build it now, in the foreground. The miner only re-sends prepare for the pair after this one.
@ -1071,11 +1194,18 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
else emit("error " + jobId + " could not build " + dir + ": " + berr);
}
}
if (!pairIs(cur, f[6], f[7])) {
if (pairIs(prepared, f[6], f[7])) {
if (!pairIsClass(cur, f[6], f[7], wantClass, wantEra)) {
char why[256];
if (pairIsClass(prepared, f[6], f[7], wantClass, wantEra)) {
if (old) releasePair(c, old);
old = cur; cur = prepared; prepared = nullptr; switched = true;
info(fmt("switched to the prepared pair epoch %.16s day %s in %.2f ms", cur->epochHex.c_str(), cur->dayHex.c_str(), wallMs() - t0));
info(fmt("switched to the prepared pair epoch %.16s day %s (class %s) in %.2f ms", cur->epochHex.c_str(), cur->dayHex.c_str(), cur->programClass.c_str(), wallMs() - t0));
} else if (pairIs(cur, f[6], f[7]) && !pf_pack_class_ok(cur->programClass.c_str(), cur->eraHex.c_str(), wantClass.c_str(), wantEra.c_str(), why, sizeof(why))) {
// Counter ASIC 2.0: right seeds, wrong class or era. The miner prepares the pair again from a pack of
// the class the chain is on (the `need` line), and the pack of the other class is never mined.
emit(fmt("need %s %s", f[6].c_str(), f[7].c_str()));
emit(fmt("error %s pack %s: %s", jobId.c_str(), cur->dir.c_str(), why));
continue;
} else if (!hexEq(cur->epochHex, f[6])) {
emit(fmt("need %s %s", f[6].c_str(), f[7].c_str())); // the miner prepares this pair (4 October 2026)
emit(fmt("error %s epoch seed mismatch: this worker holds epoch %.16s (program words %08x %08x ...)%s, the job is for epoch %.16s (bare seed words %08x %08x ...); send prepare with a pack directory",
@ -1142,11 +1272,201 @@ static int runServe(Ctx& c, const Options& o, Pair* cur) {
// ---------------------------------------------------------------------------------------------
// Main
// ---------------------------------------------------------------------------------------------
// Read-width experiment (5 October 2026, docs/plans/read-width.md): --bench and --memprobe
static uint64_t fnv1a64Bytes(const void* p, size_t n) {
const uint8_t* b = (const uint8_t*)p;
uint64_t h = 0xcbf29ce484222325ull;
for (size_t i = 0; i < n; ++i) { h ^= b[i]; h *= 0x100000001b3ull; }
return h;
}
// --bench: the pair is built and self-tested (vectors through the bound kernel with the seed words); then a warm-up
// dispatch at base nonce 0 (fingerprinted) and --batches timed dispatches of 2^B nonces, wall time around
// cuStreamSynchronize (the driver API path loads no event symbols; a 2^24 dispatch is 60 to 900 ms on the cards here,
// so the launch overhead is under 1 percent).
static int runBench(Ctx& c, const Options& o, Pair* p) {
uint32_t nonces = 1u << o.batchLog2, block = 32u * (uint32_t)c.blockWarps;
if (p->persistent) { uint32_t unit = 32u * (uint32_t)p->warps; nonces = (nonces / unit) * unit; if (nonces == 0) nonces = unit; }
CUdeviceptr dOut = 0;
std::string err;
if (c.drv.memAlloc(&dOut, (size_t)nonces * 8u) != CUDA_SUCCESS) { std::printf("FAIL: cuMemAlloc out\n"); return 2; }
std::vector<uint64_t> hOut(nonces);
double sum = 0, warm = 0;
uint64_t fp = 0;
for (int b = -1; b < o.batches; ++b) {
double t0 = wallMs();
if (!launchHash(c, p, dOut, (uint32_t)(b + 1) * nonces, p->sw, nonces, block, nullptr, err)) { std::printf("FAIL: %s\n", err.c_str()); return 2; }
CUresult r = c.drv.streamSynchronize(nullptr);
if (r != CUDA_SUCCESS) { std::printf("FAIL: dispatch %d: %s\n", b, c.err(r).c_str()); return 2; }
double ms = wallMs() - t0;
if (b < 0) {
warm = ms;
if (c.drv.memcpyDtoH(hOut.data(), dOut, (size_t)nonces * 8u) != CUDA_SUCCESS) { std::printf("FAIL: read-back\n"); return 2; }
fp = fnv1a64Bytes(hOut.data(), (size_t)nonces * 8u);
} else sum += ms;
}
c.drv.memFree(dOut);
std::string dev = c.name; for (char& ch : dev) if (ch == ' ') ch = '_';
std::printf("warm-up dispatch (base 0): %.2f ms; %d timed dispatches of %u nonces: mean %.2f ms\n", warm, o.batches, nonces, sum / o.batches);
std::printf("RESULT pack=%s class=%s device=%s arch=%s regs=%d blocks_per_sm=%d warps=%d resident=%d arena_mib=%llu hot_mib=%u hot_slots=%u hot_fill_ms=%.2f nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=wall\n",
p->dir.c_str(), p->loadClass.c_str(), dev.c_str(), c.archOpt.c_str(), p->regs, p->blocksPerSM, p->warps, p->residentWarps, (unsigned long long)(p->scratchBytes >> 20), p->hotMb, p->hotSlots, p->hotMs, nonces, o.batches,
p->checked ? (p->checkPass ? "PASS" : "FAIL") : "skipped", (unsigned long long)fp, (double)nonces * (double)o.batches / (sum / 1000.0) / 1e6,
p->loadsPerHash, p->bytesPerHash, p->scratchOps * 8u);
return 0;
}
// --memprobe: the OpenCL worker's table (proto-opencl/host.c, 5 October 2026) in CUDA C through NVRTC, so the two
// vendors are probed with the same access patterns: a dependent chain of random 4-byte reads (the hash's pattern),
// eight independent chains per lane, dependent random 16-byte and 64-byte reads, a coalesced stream and an integer
// chain, at 4, 64 and 1024 MiB. Wall time around cuStreamSynchronize, best of 3, a fresh seed per repetition.
static const char* PROBE_CUDA =
"#include <cstdint>\n"
"__device__ __forceinline__ uint32_t pm_mix(uint32_t x) { x ^= x >> 16; x *= 0x7feb352du; x ^= x >> 15; x *= 0x846ca68bu; x ^= x >> 16; return x; }\n"
"extern \"C\" __global__ void probe_fill(uint32_t* ds, uint32_t n) { uint32_t i = blockIdx.x * blockDim.x + threadIdx.x; if (i < n) ds[i] = pm_mix(i ^ 0x9E3779B9u); }\n"
"extern \"C\" __global__ void probe_chase(const uint32_t* ds, uint32_t mask, uint32_t steps, uint32_t seed, uint32_t* out) {\n"
" uint32_t g = blockIdx.x * blockDim.x + threadIdx.x; uint32_t x = pm_mix(g ^ seed);\n"
" for (uint32_t s = 0u; s < steps; ++s) x = ds[x & mask] ^ (x * 0x9E3779B1u + s);\n"
" out[g] = x;\n"
"}\n"
"extern \"C\" __global__ void probe_indep(const uint32_t* ds, uint32_t mask, uint32_t steps, uint32_t seed, uint32_t* out) {\n"
" uint32_t g = blockIdx.x * blockDim.x + threadIdx.x;\n"
" uint32_t x0 = pm_mix(g * 8u ^ seed), x1 = pm_mix((g * 8u + 1u) ^ seed), x2 = pm_mix((g * 8u + 2u) ^ seed), x3 = pm_mix((g * 8u + 3u) ^ seed);\n"
" uint32_t x4 = pm_mix((g * 8u + 4u) ^ seed), x5 = pm_mix((g * 8u + 5u) ^ seed), x6 = pm_mix((g * 8u + 6u) ^ seed), x7 = pm_mix((g * 8u + 7u) ^ seed);\n"
" for (uint32_t s = 0u; s < steps; ++s) {\n"
" x0 = ds[x0 & mask] ^ (x0 * 0x9E3779B1u + s); x1 = ds[x1 & mask] ^ (x1 * 0x9E3779B1u + s);\n"
" x2 = ds[x2 & mask] ^ (x2 * 0x9E3779B1u + s); x3 = ds[x3 & mask] ^ (x3 * 0x9E3779B1u + s);\n"
" x4 = ds[x4 & mask] ^ (x4 * 0x9E3779B1u + s); x5 = ds[x5 & mask] ^ (x5 * 0x9E3779B1u + s);\n"
" x6 = ds[x6 & mask] ^ (x6 * 0x9E3779B1u + s); x7 = ds[x7 & mask] ^ (x7 * 0x9E3779B1u + s);\n"
" }\n"
" out[g] = x0 ^ x1 ^ x2 ^ x3 ^ x4 ^ x5 ^ x6 ^ x7;\n"
"}\n"
"extern \"C\" __global__ void probe_line16(const uint4* ds, uint32_t vecMask, uint32_t steps, uint32_t seed, uint32_t* out) {\n"
" uint32_t g = blockIdx.x * blockDim.x + threadIdx.x; uint32_t x = pm_mix(g ^ seed);\n"
" for (uint32_t s = 0u; s < steps; ++s) { uint4 a = ds[x & vecMask]; x = (a.x ^ a.y ^ a.z ^ a.w) ^ (x * 0x9E3779B1u + s); }\n"
" out[g] = x;\n"
"}\n"
"extern \"C\" __global__ void probe_line(const uint4* ds, uint32_t lineMask, uint32_t steps, uint32_t seed, uint32_t* out) {\n"
" uint32_t g = blockIdx.x * blockDim.x + threadIdx.x; uint32_t x = pm_mix(g ^ seed);\n"
" for (uint32_t s = 0u; s < steps; ++s) { uint32_t l = (x & lineMask) * 4u; uint4 a = ds[l], b = ds[l + 1u], c = ds[l + 2u], d = ds[l + 3u]; x = (a.x ^ b.y ^ c.z ^ d.w) ^ (x * 0x9E3779B1u + s); }\n"
" out[g] = x;\n"
"}\n"
"extern \"C\" __global__ void probe_stream(const uint4* ds, uint32_t perLane, uint32_t* out) {\n"
" uint32_t g = blockIdx.x * blockDim.x + threadIdx.x, n = gridDim.x * blockDim.x; uint4 acc = make_uint4(0u, 0u, 0u, 0u);\n"
" for (uint32_t s = 0u; s < perLane; ++s) { uint4 v = ds[s * n + g]; acc.x ^= v.x; acc.y ^= v.y; acc.z ^= v.z; acc.w ^= v.w; }\n"
" out[g] = acc.x ^ acc.y ^ acc.z ^ acc.w;\n"
"}\n"
"extern \"C\" __global__ void probe_alu(uint32_t steps, uint32_t seed, uint32_t* out) {\n"
" uint32_t g = blockIdx.x * blockDim.x + threadIdx.x; uint32_t x = pm_mix(g ^ seed), y = x ^ 0x5bd1e995u;\n"
" for (uint32_t s = 0u; s < steps; ++s) { x = x * 0x9E3779B1u + ((y << 7u) | (y >> 25u)); y = (y ^ x) + s; }\n"
" out[g] = x ^ y;\n"
"}\n";
static double probeLaunch(Ctx& c, CUfunction f, size_t lanes, size_t local, int reps, int seedArg, uint32_t seed, void** args) {
double best = -1;
for (int r = 0; r < reps; ++r) {
uint32_t s = seed + (uint32_t)r * 0x9E3779B9u;
if (seedArg >= 0) args[seedArg] = &s;
double t0 = wallMs();
if (c.drv.launchKernel(f, (unsigned)(lanes / local), 1, 1, (unsigned)local, 1, 1, 0, nullptr, args, nullptr) != CUDA_SUCCESS) return -1;
if (c.drv.streamSynchronize(nullptr) != CUDA_SUCCESS) return -1;
double ms = wallMs() - t0;
if (best < 0 || ms < best) best = ms;
}
return best;
}
static int runMemprobe(Ctx& c, const Options& o) {
Compiled cp;
std::string err;
if (!rtcCompile(c, PROBE_CUDA, "probe.cu", "", "", {}, cp, err)) { std::printf("memprobe: build FAILED: %s\n", err.c_str()); return 2; }
CUmodule mod = nullptr;
if (c.drv.moduleLoadData(&mod, cp.image.data()) != CUDA_SUCCESS) { std::printf("memprobe: cuModuleLoadData failed\n"); return 2; }
CUfunction kFill, kChase, kIndep, kLine16, kLine, kStream, kAlu;
const char* names[7] = { "probe_fill", "probe_chase", "probe_indep", "probe_line16", "probe_line", "probe_stream", "probe_alu" };
CUfunction* fns[7] = { &kFill, &kChase, &kIndep, &kLine16, &kLine, &kStream, &kAlu };
for (int i = 0; i < 7; ++i) if (c.drv.moduleGetFunction(fns[i], mod, names[i]) != CUDA_SUCCESS) { std::printf("memprobe: %s not in the module\n", names[i]); return 2; }
int sizes[3] = { 4, 64, 1024 }, nSizes = 3;
if (o.probeMib > 0) { sizes[0] = o.probeMib; nSizes = 1; }
const size_t lanesList[8] = { 256, 1024, 1u << 12, 1u << 14, 1u << 16, 1u << 18, 1u << 20, 1u << 22 };
const size_t groups[2] = { 32, 256 };
const uint32_t STEPS = 256u, ALU_STEPS = 4096u;
const size_t maxLanes = 1u << 22;
CUdeviceptr dOut = 0;
if (c.drv.memAlloc(&dOut, maxLanes * 4u) != CUDA_SUCCESS) { std::printf("memprobe: cuMemAlloc out\n"); return 2; }
std::printf("memprobe on %s (sm_%d%d, %d SMs, driver %d.%d, NVRTC %d.%d), wall time around cuStreamSynchronize\n", c.name.c_str(), c.major, c.minor, c.sms, c.driverVersion / 1000, (c.driverVersion % 100) / 10, c.rtcMajor, c.rtcMinor);
std::printf("| probe | MiB | block | lanes in flight | steps per lane | best ms | G loads/s | ns per dependent load |\n|---|---|---|---|---|---|---|---|\n");
for (int si = 0; si < nSizes; ++si) {
int mib = sizes[si];
uint64_t bytes = (uint64_t)mib << 20;
uint32_t words = (uint32_t)(bytes / 4ull), mask = words - 1u, n = words;
CUdeviceptr dDs = 0;
if (c.drv.memAlloc(&dDs, (size_t)bytes) != CUDA_SUCCESS) { std::printf("| chase | %d | skipped: cuMemAlloc failed | | | | | |\n", mib); continue; }
{ void* a[2] = { &dDs, &n }; probeLaunch(c, kFill, ((size_t)words + 255) / 256 * 256, 256, 1, -1, 0, a); }
for (int gi = 0; gi < 2; ++gi) {
size_t local = groups[gi];
for (int li = 0; li < 8; ++li) {
size_t lanes = lanesList[li];
if (lanes < local) continue;
uint32_t seed = 0x1234567u + (uint32_t)li * 977u, steps = STEPS;
void* a[5] = { &dDs, &mask, &steps, &seed, &dOut };
double ms = probeLaunch(c, kChase, lanes, local, 3, 3, seed, a);
std::printf("| chase | %d | %zu | %zu | %u | %.3f | %.3f | %.0f |\n", mib, local, lanes, STEPS, ms, (double)lanes * STEPS / (ms / 1000.0) / 1e9, ms * 1e6 / STEPS);
std::fflush(stdout);
}
}
for (size_t lanes = 1u << 16; lanes <= maxLanes; lanes <<= 2) {
uint32_t seed = 0x7654321u, steps = STEPS;
void* a[5] = { &dDs, &mask, &steps, &seed, &dOut };
double ms = probeLaunch(c, kIndep, lanes, 256, 3, 3, seed, a);
std::printf("| indep x8 | %d | 256 | %zu | %u | %.3f | %.3f | (8 loads in flight per lane) |\n", mib, lanes, STEPS, ms, (double)lanes * 8.0 * STEPS / (ms / 1000.0) / 1e9);
}
for (size_t lanes = 1u << 14; lanes <= maxLanes; lanes <<= 2) {
uint32_t vecMask = (words / 4u) - 1u, seed = 0x2718281u, steps = STEPS;
void* a[5] = { &dDs, &vecMask, &steps, &seed, &dOut };
double ms = probeLaunch(c, kLine16, lanes, 256, 3, 3, seed, a);
std::printf("| line 16 B | %d | 256 | %zu | %u | %.3f | %.3f G reads/s | %.1f GB/s in 16 B reads |\n", mib, lanes, STEPS, ms, (double)lanes * STEPS / (ms / 1000.0) / 1e9, (double)lanes * STEPS * 16.0 / (ms / 1000.0) / 1e9);
}
for (size_t lanes = 1u << 14; lanes <= maxLanes; lanes <<= 2) {
uint32_t lineMask = (words / 16u) - 1u, seed = 0x3141592u, steps = STEPS;
void* a[5] = { &dDs, &lineMask, &steps, &seed, &dOut };
double ms = probeLaunch(c, kLine, lanes, 256, 3, 3, seed, a);
std::printf("| line 64 B | %d | 256 | %zu | %u | %.3f | %.3f G lines/s | %.1f GB/s in lines |\n", mib, lanes, STEPS, ms, (double)lanes * STEPS / (ms / 1000.0) / 1e9, (double)lanes * STEPS * 64.0 / (ms / 1000.0) / 1e9);
}
{
size_t lanes = 1u << 20;
uint32_t perLane = (uint32_t)((uint64_t)words / 4ull / (uint64_t)lanes);
if (perLane == 0) { perLane = 1; lanes = (size_t)words / 4u; }
double bytesRead = (double)perLane * (double)lanes * 16.0;
void* a[3] = { &dDs, &perLane, &dOut };
double ms = probeLaunch(c, kStream, lanes, 256, 3, -1, 0, a);
std::printf("| stream | %d | 256 | %zu | %u | %.3f | %.1f GB/s coalesced | (%.0f MiB read once) |\n", mib, lanes, perLane, ms, bytesRead / (ms / 1000.0) / 1e9, bytesRead / 1048576.0);
}
c.drv.memFree(dDs);
std::fflush(stdout);
}
{
size_t lanes = 1u << 20;
uint32_t seed = 0x2468aceu, steps = ALU_STEPS;
void* a[3] = { &steps, &seed, &dOut };
double ms = probeLaunch(c, kAlu, lanes, 256, 3, 1, seed, a);
double ops = (double)lanes * ALU_STEPS * 5.0;
std::printf("| alu | 0 | 256 | %zu | %u | %.3f | %.1f G int ops/s | %.3f G steps/s per SM (approximate: 5 ops per step counted) |\n", lanes, ALU_STEPS, ms, ops / (ms / 1000.0) / 1e9, (double)lanes * ALU_STEPS / (ms / 1000.0) / 1e9 / (c.sms ? c.sms : 1));
}
c.drv.memFree(dOut);
c.drv.moduleUnload(mod);
std::printf("memprobe: done\n");
return 0;
}
int main(int argc, char** argv) {
Options o = parseArgs(argc, argv);
Ctx c;
c.blockWarps = o.blockWarps;
c.race = o.race; c.raceBenchMs = o.raceBenchMs; c.raceBudgetS = o.raceBudgetS; c.raceRounds = o.raceRounds; c.batchLog2 = o.batchLog2; c.pinned = o.pinned;
c.warps = o.warps; c.batches = o.batches;
if (o.bench || o.memprobe) c.race = "off";
if (!o.tuningPath.empty()) { bool ok = false; c.tuning = readText(o.tuningPath, ok); if (!ok) c.tuning.clear(); }
std::string err, drvLib, rtcLib;
if (!loadDriver(c.drv, err, drvLib)) { emit("error 0 " + err); return 2; }
@ -1155,9 +1475,17 @@ int main(int argc, char** argv) {
info(fmt("igneum-worker-cuda %s: device %d %s (sm_%d%d, %d SMs), driver %d.%d from %s, NVRTC %d.%d from %s, target %s (%s)",
WORKER_VERSION, o.device, c.name.c_str(), c.major, c.minor, c.sms, c.driverVersion / 1000, (c.driverVersion % 100) / 10, drvLib.c_str(), c.rtcMajor, c.rtcMinor, rtcLib.c_str(), c.archOpt.c_str(), c.why.c_str()));
if (!c.tuning.empty()) info(fmt("tuning file %s (%zu bytes): %s", o.tuningPath.c_str(), c.tuning.size(), readTuning(c.tuning, c.name).found ? "has an entry for this card" : "no entry for this card"));
if (o.memprobe) { int rc = runMemprobe(c, o); c.drv.primaryCtxRelease(c.dev); return rc; }
double t0 = wallMs();
Pair* cur = buildPair(c, o.pack, nullptr, err, !o.check);
Pair* cur = buildPair(c, o.pack, nullptr, err, !o.check && !o.bench);
if (!cur) { emit("error 0 " + err); return 1; }
if (o.bench) {
std::printf("pack %s on %s: %s\n", o.pack.c_str(), c.name.c_str(), pairSummary(cur).c_str());
int rc = runBench(c, o, cur);
releasePair(c, cur);
c.drv.primaryCtxRelease(c.dev);
return rc;
}
if (o.raceOnly) {
std::printf("race %s on %s (%s, %d SMs, driver %d.%d, NVRTC %d.%d, %s): %s\n", o.pack.c_str(), c.name.c_str(), c.archOpt.c_str(), c.sms, c.driverVersion / 1000, (c.driverVersion % 100) / 10, c.rtcMajor, c.rtcMinor, c.why.c_str(), pairSummary(cur).c_str());
std::printf("%s\n", cur->raceLine.c_str());

View file

@ -0,0 +1,281 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 0 2 10 15 of w.
static inline uint mh_j(uint w) { return ((w >> 0u) & 1u) | (((w >> 2u) & 1u) << 1) | (((w >> 10u) & 1u) << 2) | (((w >> 15u) & 1u) << 3); }
static inline uint mh_t(uint w) { w = (w & 0x00007fffu) | ((w >> 16u) << 15u); w = (w & 0x000003ffu) | ((w >> 11u) << 10u); w = (w & 0x00000003u) | ((w >> 3u) << 2u); w = (w & 0x00000000u) | ((w >> 1u) << 0u); return w; }
static inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 0u) << 1u) | (w & 0x00000000u) | (((j >> 0u) & 1u) << 0u); w = ((w >> 2u) << 3u) | (w & 0x00000003u) | (((j >> 1u) & 1u) << 2u); w = ((w >> 10u) << 11u) | (w & 0x000003ffu) | (((j >> 2u) & 1u) << 10u); w = ((w >> 15u) << 16u) | (w & 0x00007fffu) | (((j >> 3u) & 1u) << 15u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
for (uint i = 0u; i < 16u; ++i) ds[(ulong)mh_addr(t, i)] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r6 = r6 ^ t_; } // 14 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r3 = r3 ^ t_; } // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = mul_hi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = mul_hi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r0 = r0 ^ t_; } // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = mul_hi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r5 = r5 ^ t_; } // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif

View file

@ -0,0 +1,163 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
#include "memhard.h"
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
if (t < nItems) {
uint32_t s[16];
mh_item(cache, t, s);
for (uint32_t i = 0u; i < 16u; ++i) ds[(size_t)mh_addr(t, i)] = s[i];
}
}
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint32_t x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint32_t x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint32_t x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint32_t x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint32_t x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint32_t x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint32_t x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 14 shfl
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = __umulhi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = __umulhi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = __umulhi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
if (nSegments == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nSegments + block - 1u) / block;
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
return cudaGetLastError();
}
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
if (nItems == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nItems + block - 1u) / block;
igneum_build<<<grid, block>>>(ds, cache, nItems);
return cudaGetLastError();
}
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps) {
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,375 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 0 2 10 15 of w.
static inline uint mh_j(uint w) { return ((w >> 0u) & 1u) | (((w >> 2u) & 1u) << 1) | (((w >> 10u) & 1u) << 2) | (((w >> 15u) & 1u) << 3); }
static inline uint mh_t(uint w) { w = (w & 0x00007fffu) | ((w >> 16u) << 15u); w = (w & 0x000003ffu) | ((w >> 11u) << 10u); w = (w & 0x00000003u) | ((w >> 3u) << 2u); w = (w & 0x00000000u) | ((w >> 1u) << 0u); return w; }
static inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 0u) << 1u) | (w & 0x00000000u) | (((j >> 0u) & 1u) << 0u); w = ((w >> 2u) << 3u) | (w & 0x00000003u) | (((j >> 1u) & 1u) << 2u); w = ((w >> 10u) << 11u) | (w & 0x000003ffu) | (((j >> 2u) & 1u) << 10u); w = ((w >> 15u) << 16u) | (w & 0x00007fffu) | (((j >> 3u) & 1u) << 15u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
for (uint i = 0u; i < 16u; ++i) ds[(ulong)mh_addr(t, i)] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r6 = r6 ^ t_; } // 14 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r3 = r3 ^ t_; } // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = mul_hi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = mul_hi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r0 = r0 ^ t_; } // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = mul_hi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r5 = r5 ^ t_; } // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r6 = r6 ^ t_; } // 14 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r3 = r3 ^ t_; } // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = mul_hi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = mul_hi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r0 = r0 ^ t_; } // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = mul_hi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r5 = r5 ^ t_; } // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}

View file

@ -0,0 +1,123 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
// Host declarations (also in program_bound.h if present):
// struct IgneumInitWords { uint32_t w[8]; };
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
struct IgneumInitWords { uint32_t w[8]; };
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 14 shfl
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = __umulhi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = __umulhi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = __umulhi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
return cudaGetLastError();
}
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,112 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#elif defined(_MSC_VER) && !defined(__cplusplus)
#define IGNEUM_HD static __inline
#else
#define IGNEUM_HD static inline
#endif
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint32_t r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint32_t r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 0 2 10 15 of w.
IGNEUM_HD uint32_t mh_j(uint32_t w) { return ((w >> 0u) & 1u) | (((w >> 2u) & 1u) << 1) | (((w >> 10u) & 1u) << 2) | (((w >> 15u) & 1u) << 3); }
IGNEUM_HD uint32_t mh_t(uint32_t w) { w = (w & 0x00007fffu) | ((w >> 16u) << 15u); w = (w & 0x000003ffu) | ((w >> 11u) << 10u); w = (w & 0x00000003u) | ((w >> 3u) << 2u); w = (w & 0x00000000u) | ((w >> 1u) << 0u); return w; }
IGNEUM_HD uint32_t mh_addr(uint32_t t, uint32_t j) { uint32_t w = t; w = ((w >> 0u) << 1u) | (w & 0x00000000u) | (((j >> 0u) & 1u) << 0u); w = ((w >> 2u) << 3u) | (w & 0x00000003u) | (((j >> 1u) & 1u) << 2u); w = ((w >> 10u) << 11u) | (w & 0x000003ffu) | (((j >> 2u) & 1u) << 10u); w = ((w >> 15u) << 16u) | (w & 0x00007fffu) | (((j >> 3u) & 1u) << 15u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }

View file

@ -0,0 +1,109 @@
#include <metal_stdlib>
using namespace metal;
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
inline void mh_cache_segment(device uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
inline void mh_mixer(thread uint* s, uint rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 0 2 10 15 of w.
inline uint mh_j(uint w) { return ((w >> 0u) & 1u) | (((w >> 2u) & 1u) << 1) | (((w >> 10u) & 1u) << 2) | (((w >> 15u) & 1u) << 3); }
inline uint mh_t(uint w) { w = (w & 0x00007fffu) | ((w >> 16u) << 15u); w = (w & 0x000003ffu) | ((w >> 11u) << 10u); w = (w & 0x00000003u) | ((w >> 3u) << 2u); w = (w & 0x00000000u) | ((w >> 1u) << 0u); return w; }
inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 0u) << 1u) | (w & 0x00000000u) | (((j >> 0u) & 1u) << 0u); w = ((w >> 2u) << 3u) | (w & 0x00000003u) | (((j >> 1u) & 1u) << 2u); w = ((w >> 10u) << 11u) | (w & 0x000003ffu) | (((j >> 2u) & 1u) << 10u); w = ((w >> 15u) << 16u) | (w & 0x00007fffu) | (((j >> 3u) & 1u) << 15u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
// One thread per segment (2^16 threads).
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
mh_cache_segment(cache, gid);
}
// One thread per 64-byte item (dataset words / 16 threads).
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
uint s[16];
mh_item(cache, gid, s);
for (uint i = 0u; i < 16u; ++i) dataset[mh_addr(gid, i)] = s[i];
}

View file

@ -0,0 +1,76 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#ifndef IGNEUM_NO_CUDA
#include <cuda_runtime.h>
#endif
#define IGNEUM_SEED_STRING "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000"
#define IGNEUM_SEED_BYTES_HEX "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"
#define IGNEUM_GENERATOR 3
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0x73bcbfe8ccf988f1ull
#define IGNEUM_DAY_STRING "bytes:69676e65756d2d6461792ffa50000000000000"
#define IGNEUM_DAY_BYTES_HEX "69676e65756d2d6461792ffa50000000000000"
#define IGNEUM_DAY0 0xceed56d7u
#define IGNEUM_DAY1 0x9ba270d2u
#define IGNEUM_DATASET_LOG2 28
#define IGNEUM_MASK 0x0fffffffu
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS 8
#define IGNEUM_INSTR_COUNT 64
#define IGNEUM_LOADS_PER_HASH 128
#define IGNEUM_WIDE_LOADS_PER_HASH 0
#define IGNEUM_OP_MIX "load=16 add=15 shfl=6 mad=4 or=4 rotl=4 rotr=4 xor=4 mulhi=3 mul=2 sub=2"
// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that
// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=<hex>).
#define IGNEUM_PROGRAM_CLASS "v3"
#define IGNEUM_ERA_SEED_HEX "8bffdd3366b9c3ffe89c1231e91538d531cfa717307df5c48ccba43096e17212"
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
#define IGNEUM_LOAD_CLASS "w4-erab2ed8a89"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 512
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Era layout (5 October 2026, docs/plans/era-layout.md): NOT the lottery hash. Every dataset load reads
// idx = ((rotl(src * STRIDE_MUL, STRIDE_ROT) & window mask) | window offset) & MASK; the window of a load site is the
// dataset, a half or a quarter of it (IGNEUM_ERA_WINDOWS: site:shrink:offset); dataset word w holds word j(w) of item
// t(w) with j's bits at the INTERLEAVE positions (memhard.h: mh_t, mh_j, mh_addr).
#define IGNEUM_ERA_LABEL "b2ed8a89"
#define IGNEUM_ERA_SEED_WORDS { 0xb2ed8a89u, 0xb023f2bau, 0x6bbf405eu, 0x98153cddu, 0x49428e54u, 0xfbfa65eeu, 0xbcb76d0du, 0x2f509891u }
#define IGNEUM_ERA_ALLOWED_WIDTHS { 1, 0, 0 } // words, ascending, 0 = unused; one entry pins the width
#define IGNEUM_ERA_WIDTH_WORDS 1
#define IGNEUM_ERA_STRIDE_MUL 0x625e5ab3u
#define IGNEUM_ERA_STRIDE_ROT 19
#define IGNEUM_ERA_INTERLEAVE { 0, 2, 10, 15 }
#define IGNEUM_ERA_WINDOWS "7:2:1 8:1:1 9:1:1 10:1:1 11:0:0 13:1:1 29:0:0 30:2:2 31:1:1 44:1:1 46:2:0 47:0:0 52:0:0 56:0:0 58:2:0 63:1:1"
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1
#define IGNEUM_SEEDW_INIT { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u }
#define IGNEUM_KEY_INIT { 0xceed56d7u, 0x9ba270d2u, 0x82caab2du, 0x81ebce0eu, 0x12b6ecf1u, 0xd0f3fd7cu, 0xd872eefeu, 0xc158c7bdu }
#define IGNEUM_CACHE_LOG2_WORDS 26
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
#define IGNEUM_CACHE_SEGMENTS 65536u
#define IGNEUM_ITEM_ROUNDS 8
#define IGNEUM_MIX_ROT_INIT { 17u, 12u, 20u, 23u, 7u, 3u, 27u, 16u }
#define IGNEUM_MIX_MUL_INIT { 0xf351d601u, 0xa3bb398fu, 0xb5a09e35u, 0x7509c9c1u, 0x6bbf31e9u, 0xfc849a79u, 0xded91851u, 0x8d9113d1u, 0x0ff15225u, 0x3a5bdd41u, 0xab533435u, 0xe1c55ad5u, 0xe6d3bd0du, 0x9d9ffbbdu, 0xbb2a3cf3u, 0x50a7c08du }
#define IGNEUM_MIX_RC_INIT { 0xc6892460u, 0x25b7228au, 0xcd515004u, 0x2846527au, 0xa6324241u, 0x36e3ec53u, 0x82961bacu, 0x0f97ba7du, 0xb6f921a9u, 0x3ada24e5u, 0xde20ab91u, 0x5378eeb2u, 0x7d161662u, 0x89353cc1u, 0xb1aa03a2u, 0x788acae6u }
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#endif

View file

@ -0,0 +1,143 @@
{
"format": "igneum-program-pack-3",
"generator": 3,
"attempt": 0,
"program_id": "0x73bcbfe8ccf988f1",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000",
"seed_bytes": "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07",
"seed_words": ["0x667d0fbd", "0x7b8e5963", "0x31c67e5e", "0x4529ddc6", "0xef19d6d8", "0xaccf6211", "0xda0aed32", "0xabc6df31"],
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
"lanes": 32,
"registers": 8,
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"program_class": "v3",
"era_seed_bytes": "8bffdd3366b9c3ffe89c1231e91538d531cfa717307df5c48ccba43096e17212",
"load_class": "w4-erab2ed8a89",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [16, 0, 0],
"bytes_per_hash": 512,
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"era": {
"label": "b2ed8a89",
"seed_words": ["0xb2ed8a89", "0xb023f2ba", "0x6bbf405e", "0x98153cdd", "0x49428e54", "0xfbfa65ee", "0xbcb76d0d", "0x2f509891"],
"draw": "docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used",
"allowed_widths": [1],
"width_words": 1,
"stride_mul": "0x625e5ab3",
"stride_rot": 19,
"interleave": [0, 2, 10, 15],
"address": "y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words",
"windows": "per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)",
"dataset_word": "dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed",
"program_id_suffix": "'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]"
},
"op_mix": {"load": 16, "add": 15, "shfl": 6, "mad": 4, "or": 4, "rotl": 4, "rotr": 4, "xor": 4, "mulhi": 3, "mul": 2, "sub": 2},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
"op_semantics": {
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
"sub": "dst = dst - src",
"mul": "dst = dst * src (low 32)",
"mulhi": "dst = high 32 bits of dst * src",
"xor": "dst = dst ^ src",
"or": "dst = dst | src",
"rotl": "dst = rotl(dst, rot), rot in 1..31",
"rotr": "dst = rotr(dst, src & 31)",
"mad": "dst = src * src2 + dst",
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
"load": "dst = dst ^ dataset[src & dataset.mask]",
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
},
"dataset": {
"log2_words": 28,
"bytes": 1073741824,
"mask": "0x0fffffff",
"day": "bytes:69676e65756d2d6461792ffa50000000000000",
"day_bytes": "69676e65756d2d6461792ffa50000000000000",
"day_words_from": "seed_words_from_bytes(day_bytes)",
"d0": "0xceed56d7",
"d1": "0x9ba270d2",
"mode": "memory-hard",
"spec": "proto-metal/MEMHARD.md",
"key": ["0xceed56d7", "0x9ba270d2", "0x82caab2d", "0x81ebce0e", "0x12b6ecf1", "0xd0f3fd7c", "0xd872eefe", "0xc158c7bd"],
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [17, 12, 20, 23, 7, 3, 27, 16], "mul": ["0xf351d601", "0xa3bb398f", "0xb5a09e35", "0x7509c9c1", "0x6bbf31e9", "0xfc849a79", "0xded91851", "0x8d9113d1", "0x0ff15225", "0x3a5bdd41", "0xab533435", "0xe1c55ad5", "0xe6d3bd0d", "0x9d9ffbbd", "0xbb2a3cf3", "0x50a7c08d"], "rc": ["0xc6892460", "0x25b7228a", "0xcd515004", "0x2846527a", "0xa6324241", "0x36e3ec53", "0x82961bac", "0x0f97ba7d", "0xb6f921a9", "0x3ada24e5", "0xde20ab91", "0x5378eeb2", "0x7d161662", "0x89353cc1", "0xb1aa03a2", "0x788acae6"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
"word": "dataset[w] = item(w >> 4)[w & 15]"
},
"instructions": [
{"i": 0, "op": "add", "dst": 4, "src": 5, "src2": 7, "imm": "0xea86e152", "imm2": "0x5810667a", "rot": 27, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 1, "op": "xor", "dst": 7, "src": 0, "src2": 6, "imm": "0xbaab6229", "imm2": "0xed861989", "rot": 26, "bit": 22, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 2, "op": "shfl", "dst": 3, "src": 6, "src2": 2, "imm": "0x5b623116", "imm2": "0xff12e5b2", "rot": 12, "bit": 24, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 3, "op": "sub", "dst": 4, "src": 1, "src2": 1, "imm": "0xe99741c7", "imm2": "0xf5fa5009", "rot": 1, "bit": 21, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 4, "op": "mad", "dst": 2, "src": 0, "src2": 4, "imm": "0x673c2157", "imm2": "0xee02465f", "rot": 20, "bit": 22, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 5, "op": "rotl", "dst": 4, "src": 7, "src2": 7, "imm": "0x946f7818", "imm2": "0x45d3399e", "rot": 9, "bit": 2, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 6, "op": "rotr", "dst": 0, "src": 2, "src2": 4, "imm": "0x5f6a0ed2", "imm2": "0x7043a636", "rot": 19, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 7, "op": "load", "dst": 6, "src": 7, "src2": 1, "imm": "0x5c61dcf7", "imm2": "0x7466aa40", "rot": 19, "bit": 9, "mask": 2, "width": 1, "win": 2, "off": 1},
{"i": 8, "op": "load", "dst": 1, "src": 4, "src2": 5, "imm": "0x85668475", "imm2": "0xdb8cc483", "rot": 29, "bit": 7, "mask": 4, "width": 1, "win": 1, "off": 1},
{"i": 9, "op": "load", "dst": 1, "src": 2, "src2": 4, "imm": "0x4454980f", "imm2": "0xebd31581", "rot": 10, "bit": 28, "mask": 16, "width": 1, "win": 1, "off": 1},
{"i": 10, "op": "load", "dst": 7, "src": 0, "src2": 2, "imm": "0xe075297c", "imm2": "0x5779c44c", "rot": 10, "bit": 22, "mask": 2, "width": 1, "win": 1, "off": 1},
{"i": 11, "op": "load", "dst": 7, "src": 1, "src2": 6, "imm": "0x65aa4311", "imm2": "0x4fe48ea9", "rot": 15, "bit": 9, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 12, "op": "mul", "dst": 5, "src": 4, "src2": 2, "imm": "0x1383d3ad", "imm2": "0xf3094b29", "rot": 8, "bit": 9, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 13, "op": "load", "dst": 7, "src": 6, "src2": 2, "imm": "0xed8a496f", "imm2": "0x3072c3c6", "rot": 28, "bit": 19, "mask": 8, "width": 1, "win": 1, "off": 1},
{"i": 14, "op": "shfl", "dst": 6, "src": 5, "src2": 0, "imm": "0x8b965b57", "imm2": "0xcfeca6c1", "rot": 12, "bit": 27, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 15, "op": "shfl", "dst": 3, "src": 5, "src2": 3, "imm": "0x877c7586", "imm2": "0xa9cb2a03", "rot": 2, "bit": 29, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 16, "op": "or", "dst": 2, "src": 7, "src2": 3, "imm": "0xb740221a", "imm2": "0x89d38d6d", "rot": 6, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 17, "op": "rotr", "dst": 0, "src": 6, "src2": 1, "imm": "0x26f3ad8a", "imm2": "0x27256f15", "rot": 18, "bit": 5, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 18, "op": "add", "dst": 7, "src": 5, "src2": 7, "imm": "0xf66e7017", "imm2": "0xb9e3577e", "rot": 9, "bit": 12, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 19, "op": "rotl", "dst": 7, "src": 0, "src2": 5, "imm": "0x849ae6ee", "imm2": "0x02b358f9", "rot": 24, "bit": 18, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 20, "op": "add", "dst": 6, "src": 7, "src2": 6, "imm": "0x52334d12", "imm2": "0x8c9f0ef8", "rot": 11, "bit": 23, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 21, "op": "or", "dst": 2, "src": 6, "src2": 5, "imm": "0xb1871e63", "imm2": "0xb2e40191", "rot": 5, "bit": 13, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 22, "op": "mad", "dst": 6, "src": 5, "src2": 4, "imm": "0x97df29e4", "imm2": "0xe60fea84", "rot": 11, "bit": 16, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 23, "op": "or", "dst": 1, "src": 0, "src2": 3, "imm": "0x8f30d21d", "imm2": "0x2df685a0", "rot": 31, "bit": 0, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 24, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0xd2c4025f", "imm2": "0x5269eb4d", "rot": 31, "bit": 21, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 25, "op": "add", "dst": 2, "src": 6, "src2": 5, "imm": "0x659fc3d3", "imm2": "0x9cec0e12", "rot": 6, "bit": 17, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 26, "op": "rotr", "dst": 7, "src": 0, "src2": 1, "imm": "0xd89ef484", "imm2": "0x20be3846", "rot": 12, "bit": 9, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 27, "op": "mad", "dst": 4, "src": 5, "src2": 7, "imm": "0xb48420ae", "imm2": "0x3d1f2485", "rot": 14, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 28, "op": "mad", "dst": 3, "src": 2, "src2": 4, "imm": "0xc5c46d76", "imm2": "0x700044b5", "rot": 22, "bit": 10, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 29, "op": "load", "dst": 1, "src": 4, "src2": 3, "imm": "0x0fbaf177", "imm2": "0xfff4f2ed", "rot": 20, "bit": 31, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 30, "op": "load", "dst": 2, "src": 3, "src2": 0, "imm": "0xe90eb2e5", "imm2": "0xb0f9eb79", "rot": 27, "bit": 14, "mask": 4, "width": 1, "win": 2, "off": 2},
{"i": 31, "op": "load", "dst": 1, "src": 5, "src2": 2, "imm": "0x97ba3fc3", "imm2": "0x7894e657", "rot": 3, "bit": 30, "mask": 16, "width": 1, "win": 1, "off": 1},
{"i": 32, "op": "add", "dst": 7, "src": 2, "src2": 7, "imm": "0x070888a8", "imm2": "0xe403240e", "rot": 2, "bit": 16, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 33, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0xf2e46d55", "imm2": "0x29701828", "rot": 31, "bit": 28, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 34, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x58f75b87", "imm2": "0x343b7aee", "rot": 12, "bit": 14, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 35, "op": "mulhi", "dst": 2, "src": 5, "src2": 6, "imm": "0x5892a9e6", "imm2": "0xc9824c94", "rot": 19, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 36, "op": "xor", "dst": 4, "src": 2, "src2": 3, "imm": "0x7b5b5474", "imm2": "0x45cfc5dd", "rot": 17, "bit": 18, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 37, "op": "mul", "dst": 6, "src": 5, "src2": 7, "imm": "0xccf564a5", "imm2": "0x873ad101", "rot": 7, "bit": 11, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 38, "op": "xor", "dst": 7, "src": 0, "src2": 6, "imm": "0xcac8f06d", "imm2": "0x6b97c683", "rot": 18, "bit": 28, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 39, "op": "add", "dst": 7, "src": 2, "src2": 4, "imm": "0xb8180e9d", "imm2": "0x32bbd117", "rot": 23, "bit": 19, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 40, "op": "rotr", "dst": 2, "src": 3, "src2": 0, "imm": "0x2d6070bc", "imm2": "0x68ff101e", "rot": 13, "bit": 18, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 41, "op": "sub", "dst": 7, "src": 0, "src2": 2, "imm": "0x0daf96ea", "imm2": "0x36f37be1", "rot": 5, "bit": 0, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 42, "op": "add", "dst": 4, "src": 3, "src2": 6, "imm": "0x6ced15b7", "imm2": "0x6df7aed4", "rot": 19, "bit": 4, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 43, "op": "shfl", "dst": 7, "src": 3, "src2": 5, "imm": "0x8ace05f3", "imm2": "0xd378ec12", "rot": 23, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 44, "op": "load", "dst": 0, "src": 7, "src2": 4, "imm": "0xb0607786", "imm2": "0xc4acabbc", "rot": 13, "bit": 7, "mask": 16, "width": 1, "win": 1, "off": 1},
{"i": 45, "op": "add", "dst": 1, "src": 6, "src2": 5, "imm": "0xe09f54e9", "imm2": "0x83e825bf", "rot": 23, "bit": 14, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 46, "op": "load", "dst": 3, "src": 1, "src2": 0, "imm": "0x63cc1e4e", "imm2": "0xa1be8118", "rot": 12, "bit": 6, "mask": 2, "width": 1, "win": 2, "off": 0},
{"i": 47, "op": "load", "dst": 6, "src": 3, "src2": 7, "imm": "0x353f1d79", "imm2": "0x3b2e7456", "rot": 18, "bit": 18, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 48, "op": "mulhi", "dst": 4, "src": 2, "src2": 7, "imm": "0x00d8a3cd", "imm2": "0x231866d2", "rot": 21, "bit": 20, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 49, "op": "add", "dst": 5, "src": 0, "src2": 2, "imm": "0xa8bae6df", "imm2": "0xf572bdb9", "rot": 14, "bit": 7, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 50, "op": "shfl", "dst": 0, "src": 7, "src2": 7, "imm": "0x81ef22e1", "imm2": "0x74438fc5", "rot": 28, "bit": 18, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 51, "op": "add", "dst": 6, "src": 0, "src2": 6, "imm": "0x383b9260", "imm2": "0x11e17c61", "rot": 12, "bit": 19, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 52, "op": "load", "dst": 5, "src": 2, "src2": 2, "imm": "0xfb84f451", "imm2": "0x11cd863e", "rot": 21, "bit": 20, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 53, "op": "rotl", "dst": 6, "src": 5, "src2": 4, "imm": "0xb1a7db6b", "imm2": "0x76686b9b", "rot": 12, "bit": 4, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 54, "op": "rotl", "dst": 3, "src": 6, "src2": 3, "imm": "0x6f981f52", "imm2": "0xd99aeba2", "rot": 12, "bit": 27, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 55, "op": "add", "dst": 2, "src": 1, "src2": 2, "imm": "0xac6be8e3", "imm2": "0x18d67dbb", "rot": 26, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 56, "op": "load", "dst": 1, "src": 4, "src2": 0, "imm": "0x7e7f6a00", "imm2": "0x6f0747da", "rot": 25, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 57, "op": "add", "dst": 5, "src": 0, "src2": 4, "imm": "0xf03673fe", "imm2": "0xa75cd60d", "rot": 16, "bit": 12, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 58, "op": "load", "dst": 5, "src": 0, "src2": 2, "imm": "0x227f94a6", "imm2": "0x0e8344f9", "rot": 20, "bit": 10, "mask": 2, "width": 1, "win": 2, "off": 0},
{"i": 59, "op": "add", "dst": 1, "src": 4, "src2": 3, "imm": "0xdecd4794", "imm2": "0x8dfb96bb", "rot": 21, "bit": 7, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 60, "op": "mulhi", "dst": 3, "src": 2, "src2": 2, "imm": "0x0dd268e0", "imm2": "0x53034ca9", "rot": 1, "bit": 8, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 61, "op": "or", "dst": 6, "src": 4, "src2": 7, "imm": "0x3a45a321", "imm2": "0x9bc59a5f", "rot": 25, "bit": 11, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 62, "op": "shfl", "dst": 5, "src": 4, "src2": 2, "imm": "0x8f229cc1", "imm2": "0xcaac64a2", "rot": 17, "bit": 13, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 63, "op": "load", "dst": 3, "src": 6, "src2": 6, "imm": "0xb2574178", "imm2": "0xcbcc798d", "rot": 28, "bit": 0, "mask": 16, "width": 1, "win": 1, "off": 1}
]
}

View file

@ -0,0 +1,109 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
uint gid [[thread_position_in_grid]]) {
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + select(0xea86e152u, 0x5810667au, ((sel >> 13u) & 1u) != 0u); // 0
r7 = r7 ^ r0; // 1
r3 = r3 ^ simd_shuffle_xor(r6, (ushort)1); // 2
r4 = r4 - r1; // 3
r2 = r0 * r4 + r2; // 4
r4 = rotl_imm(r4, 9u); // 5
r0 = rotr_var(r0, r2); // 6
r6 = r6 ^ dataset[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & MASK]; // 7
r1 = r1 ^ dataset[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 8
r1 = r1 ^ dataset[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 9
r7 = r7 ^ dataset[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 10
r7 = r7 ^ dataset[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 11
r5 = r5 * r4; // 12
r7 = r7 ^ dataset[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 13
r6 = r6 ^ simd_shuffle_xor(r5, (ushort)16); // 14
r3 = r3 ^ simd_shuffle_xor(r5, (ushort)1); // 15
r2 = r2 | r7; // 16
r0 = rotr_var(r0, r6); // 17
r7 = r7 + r5 + select(0xf66e7017u, 0xb9e3577eu, ((sel >> 12u) & 1u) != 0u); // 18
r7 = rotl_imm(r7, 24u); // 19
r6 = r6 + r7 + select(0x52334d12u, 0x8c9f0ef8u, ((sel >> 23u) & 1u) != 0u); // 20
r2 = r2 | r6; // 21
r6 = r5 * r4 + r6; // 22
r1 = r1 | r0; // 23
r6 = r6 ^ r1; // 24
r2 = r2 + r6 + select(0x659fc3d3u, 0x9cec0e12u, ((sel >> 17u) & 1u) != 0u); // 25
r7 = rotr_var(r7, r0); // 26
r4 = r5 * r7 + r4; // 27
r3 = r2 * r4 + r3; // 28
r1 = r1 ^ dataset[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 29
r2 = r2 ^ dataset[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & MASK]; // 30
r1 = r1 ^ dataset[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 31
r7 = r7 + r2 + select(0x070888a8u, 0xe403240eu, ((sel >> 16u) & 1u) != 0u); // 32
r2 = r2 + r0 + select(0xf2e46d55u, 0x29701828u, ((sel >> 28u) & 1u) != 0u); // 33
r2 = r2 + r3 + select(0x58f75b87u, 0x343b7aeeu, ((sel >> 14u) & 1u) != 0u); // 34
r2 = mulhi(r2, r5); // 35
r4 = r4 ^ r2; // 36
r6 = r6 * r5; // 37
r7 = r7 ^ r0; // 38
r7 = r7 + r2 + select(0xb8180e9du, 0x32bbd117u, ((sel >> 19u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r3); // 40
r7 = r7 - r0; // 41
r4 = r4 + r3 + select(0x6ced15b7u, 0x6df7aed4u, ((sel >> 4u) & 1u) != 0u); // 42
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 43
r0 = r0 ^ dataset[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 44
r1 = r1 + r6 + select(0xe09f54e9u, 0x83e825bfu, ((sel >> 14u) & 1u) != 0u); // 45
r3 = r3 ^ dataset[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & MASK]; // 46
r6 = r6 ^ dataset[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 47
r4 = mulhi(r4, r2); // 48
r5 = r5 + r0 + select(0xa8bae6dfu, 0xf572bdb9u, ((sel >> 7u) & 1u) != 0u); // 49
r0 = r0 ^ simd_shuffle_xor(r7, (ushort)4); // 50
r6 = r6 + r0 + select(0x383b9260u, 0x11e17c61u, ((sel >> 19u) & 1u) != 0u); // 51
r5 = r5 ^ dataset[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 52
r6 = rotl_imm(r6, 12u); // 53
r3 = rotl_imm(r3, 12u); // 54
r2 = r2 + r1 + select(0xac6be8e3u, 0x18d67dbbu, ((sel >> 10u) & 1u) != 0u); // 55
r1 = r1 ^ dataset[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 56
r5 = r5 + r0 + select(0xf03673feu, 0xa75cd60du, ((sel >> 12u) & 1u) != 0u); // 57
r5 = r5 ^ dataset[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & MASK]; // 58
r1 = r1 + r4 + select(0xdecd4794u, 0x8dfb96bbu, ((sel >> 7u) & 1u) != 0u); // 59
r3 = mulhi(r3, r2); // 60
r6 = r6 | r4; // 61
r5 = r5 ^ simd_shuffle_xor(r4, (ushort)8); // 62
r3 = r3 ^ dataset[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}

View file

@ -0,0 +1,111 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
constant uint* initw [[buffer(3)]],
uint gid [[thread_position_in_grid]]) {
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + select(0xea86e152u, 0x5810667au, ((sel >> 13u) & 1u) != 0u); // 0
r7 = r7 ^ r0; // 1
r3 = r3 ^ simd_shuffle_xor(r6, (ushort)1); // 2
r4 = r4 - r1; // 3
r2 = r0 * r4 + r2; // 4
r4 = rotl_imm(r4, 9u); // 5
r0 = rotr_var(r0, r2); // 6
r6 = r6 ^ dataset[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x04000000u) & MASK]; // 7
r1 = r1 ^ dataset[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 8
r1 = r1 ^ dataset[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 9
r7 = r7 ^ dataset[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 10
r7 = r7 ^ dataset[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 11
r5 = r5 * r4; // 12
r7 = r7 ^ dataset[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 13
r6 = r6 ^ simd_shuffle_xor(r5, (ushort)16); // 14
r3 = r3 ^ simd_shuffle_xor(r5, (ushort)1); // 15
r2 = r2 | r7; // 16
r0 = rotr_var(r0, r6); // 17
r7 = r7 + r5 + select(0xf66e7017u, 0xb9e3577eu, ((sel >> 12u) & 1u) != 0u); // 18
r7 = rotl_imm(r7, 24u); // 19
r6 = r6 + r7 + select(0x52334d12u, 0x8c9f0ef8u, ((sel >> 23u) & 1u) != 0u); // 20
r2 = r2 | r6; // 21
r6 = r5 * r4 + r6; // 22
r1 = r1 | r0; // 23
r6 = r6 ^ r1; // 24
r2 = r2 + r6 + select(0x659fc3d3u, 0x9cec0e12u, ((sel >> 17u) & 1u) != 0u); // 25
r7 = rotr_var(r7, r0); // 26
r4 = r5 * r7 + r4; // 27
r3 = r2 * r4 + r3; // 28
r1 = r1 ^ dataset[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 29
r2 = r2 ^ dataset[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x08000000u) & MASK]; // 30
r1 = r1 ^ dataset[((rotl_imm(r5 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 31
r7 = r7 + r2 + select(0x070888a8u, 0xe403240eu, ((sel >> 16u) & 1u) != 0u); // 32
r2 = r2 + r0 + select(0xf2e46d55u, 0x29701828u, ((sel >> 28u) & 1u) != 0u); // 33
r2 = r2 + r3 + select(0x58f75b87u, 0x343b7aeeu, ((sel >> 14u) & 1u) != 0u); // 34
r2 = mulhi(r2, r5); // 35
r4 = r4 ^ r2; // 36
r6 = r6 * r5; // 37
r7 = r7 ^ r0; // 38
r7 = r7 + r2 + select(0xb8180e9du, 0x32bbd117u, ((sel >> 19u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r3); // 40
r7 = r7 - r0; // 41
r4 = r4 + r3 + select(0x6ced15b7u, 0x6df7aed4u, ((sel >> 4u) & 1u) != 0u); // 42
r7 = r7 ^ simd_shuffle_xor(r3, (ushort)4); // 43
r0 = r0 ^ dataset[((rotl_imm(r7 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 44
r1 = r1 + r6 + select(0xe09f54e9u, 0x83e825bfu, ((sel >> 14u) & 1u) != 0u); // 45
r3 = r3 ^ dataset[((rotl_imm(r1 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & MASK]; // 46
r6 = r6 ^ dataset[((rotl_imm(r3 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 47
r4 = mulhi(r4, r2); // 48
r5 = r5 + r0 + select(0xa8bae6dfu, 0xf572bdb9u, ((sel >> 7u) & 1u) != 0u); // 49
r0 = r0 ^ simd_shuffle_xor(r7, (ushort)4); // 50
r6 = r6 + r0 + select(0x383b9260u, 0x11e17c61u, ((sel >> 19u) & 1u) != 0u); // 51
r5 = r5 ^ dataset[((rotl_imm(r2 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 52
r6 = rotl_imm(r6, 12u); // 53
r3 = rotl_imm(r3, 12u); // 54
r2 = r2 + r1 + select(0xac6be8e3u, 0x18d67dbbu, ((sel >> 10u) & 1u) != 0u); // 55
r1 = r1 ^ dataset[((rotl_imm(r4 * 0x625e5ab3u, 19u) & 0x0fffffffu) | 0x00000000u) & MASK]; // 56
r5 = r5 + r0 + select(0xf03673feu, 0xa75cd60du, ((sel >> 12u) & 1u) != 0u); // 57
r5 = r5 ^ dataset[((rotl_imm(r0 * 0x625e5ab3u, 19u) & 0x03ffffffu) | 0x00000000u) & MASK]; // 58
r1 = r1 + r4 + select(0xdecd4794u, 0x8dfb96bbu, ((sel >> 7u) & 1u) != 0u); // 59
r3 = mulhi(r3, r2); // 60
r6 = r6 | r4; // 61
r5 = r5 ^ simd_shuffle_xor(r4, (ushort)8); // 62
r3 = r3 ^ dataset[((rotl_imm(r6 * 0x625e5ab3u, 19u) & 0x07ffffffu) | 0x08000000u) & MASK]; // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}

View file

@ -0,0 +1,5 @@
epoch_seed_hex edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07
day_seed_hex 69676e65756d2d6461792ffa50000000000000
epoch_index 0
day_index 20730
daa_score 0

View file

@ -0,0 +1,57 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#define IGNEUM_VEC_WARPS 3
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0xa32ad2dd06264089ull, 0x854fea26361763aaull, 0xd958908d36ee63b5ull, 0x33641fe988c44ad9ull, 0xb5290b7892dbc7baull, 0xd815f329e6f806c0ull, 0x9fb28ff3fc29e39cull, 0xdbc28a5d8bf42ab4ull,
0x06f955808ff71bb9ull, 0x1ec10fff7bcdeaddull, 0x1acbc289cddd83cdull, 0x395e1564cda10137ull, 0x8a2aac2aa5b190d2ull, 0x499b8401e1f3e412ull, 0x9e2320319c85fc43ull, 0x66ba9661bc421ad6ull,
0x3ba69971f1c9d8bdull, 0x00d7a64d460bfefeull, 0xc0f6487a1fbc5978ull, 0x1a261ac8673e1146ull, 0xf5a95a7caaaafa0aull, 0x7bbdabd107842744ull, 0x49dd3bf9b77a16cfull, 0xf7026cc441495a05ull,
0x49fc48c1cba2be91ull, 0x84c2fc11de378cf9ull, 0x96de9e40c9da551cull, 0x3d20f01e363aaf61ull, 0xef36922a5fcf96caull, 0x31f1b0c4b0b42aafull, 0x33c6406d516d5992ull, 0x847af3a4248a972full
},
{ // base nonce 4096
0x2039f40a61003341ull, 0x8238174df3761142ull, 0x32a72a04c4ca4a3cull, 0x16aa0fd3f65da6b3ull, 0x502bf9db9bd98782ull, 0x109f60bc6167ff30ull, 0x3873c59792d6b552ull, 0xea98fba68fc32149ull,
0x56a8440169cb0f83ull, 0x7362edc3a7c98252ull, 0x8f393a7ad9f4ff40ull, 0x6b3cb3a1e0b02453ull, 0xdcbdc40aedfa06c0ull, 0x2bc381227147a2f3ull, 0x1570ef95b0c8e31full, 0xf3ca64d0f9ca8c91ull,
0x1c767a63cd6a0bf8ull, 0x8e1a13a1237a948bull, 0xff2241b0b53b3efdull, 0x95949c528978b1fbull, 0xce3519499b78db7eull, 0x56a0fcecaaa62096ull, 0x0a62c2bddc7041b2ull, 0x5ec463510d31d7c0ull,
0x2fc74273ed47b17full, 0xedc5740b0c5c191eull, 0xe62f737c106216cbull, 0x185df496808afd70ull, 0x29070790e64a0bc7ull, 0xc0ddaeac8b22b14bull, 0x302982837b6d96b4ull, 0x523972727aa906b5ull
},
{ // base nonce 1000000
0x3527292a5f4afb4eull, 0x8c4128a03d946030ull, 0xb5b598dd18eb207dull, 0x9f1b21b5bf9dce8full, 0xa791a53fa1a4733eull, 0x12d2f9629aaca3b8ull, 0x400e844a00ba254cull, 0x1e9f98bee0272859ull,
0x9983064542aabb54ull, 0x017873e17fab719cull, 0xf3e747e327968d2bull, 0x562380f8016b9ca7ull, 0x3f5007c1ec21b4a8ull, 0x5612489676e34c2bull, 0x97e13b52bc125dabull, 0xe15cf6d8d12f9b00ull,
0xd8ba508db7b12919ull, 0x2e16d66db0bd1837ull, 0x0026e228553d7b18ull, 0x0c6d22f5018958ddull, 0xe76da566fc36580dull, 0x7824813a77faff44ull, 0x663811c5504ff04dull, 0x75efd5493353d65aull,
0x926b72ebe963caa2ull, 0x392723945438b76dull, 0x4df61663eb2bfe05ull, 0x9a305821ba9ceb59ull, 0x317f4a6ac3f65a88ull, 0xdbae9b884f6829a9ull, 0x97cc8c8866b1a0cbull, 0x5db142b6d7777df1ull
}
};
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
static const uint32_t IGNEUM_DS_HEAD[16] = {
0x3dd50b1fu, 0x48edec90u, 0x93123701u, 0x3a7d2407u, 0x5119540eu, 0x657ac748u, 0x3358ef2du, 0x09e2105cu,
0xcdd88b45u, 0xb03a0ac0u, 0x2c37c6b0u, 0x10e2c6a5u, 0x25e2929cu, 0x4317cabdu, 0x5fa0cdccu, 0x6178ee52u
};
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
static const uint32_t IGNEUM_DS_LAST = 0xf7b7180eu;
// 64 sampled dataset words (index, value) computed on the Mac.
#define IGNEUM_DS_SAMPLES 64
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
};
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
0x1faef64au, 0x068e4a54u, 0x24551c62u, 0x1234f539u, 0xc15100f5u, 0x2f53b6b0u, 0xbab5fb76u, 0x06edf897u, 0xb0880fc6u, 0xccfddc4du, 0xf640c0f4u, 0xf145a054u, 0xa2425e8fu, 0x0a218118u, 0xd310512au, 0xb612d7ffu, 0x9a290b4du, 0xaf0227edu, 0x81b9c33cu, 0x5f352daeu, 0x46889c3du, 0x8ddc5d17u, 0x368b4a6cu, 0x16acb57au, 0x103f86c1u, 0x7144efafu, 0x87cd0dc1u, 0x1d9132b1u, 0x28b0d87cu, 0x01c44610u, 0x908b6bacu, 0x9e127820u, 0x9e695a6fu, 0xfbb80109u, 0x39ad9ba2u, 0x0c867f51u, 0xac1e69e5u, 0xf16f763du, 0x2de60ff0u, 0xfabaf858u, 0x53a35511u, 0x249e753fu, 0x88f84300u, 0xd79db283u, 0x7ac395cau, 0x5df23d8bu, 0xe3f98318u, 0x33ad1217u, 0x2dd19e3fu, 0xf1a97b54u, 0xb1fabd1eu, 0x1bc27b43u, 0xb208c2e5u, 0x1f109813u, 0xccb6a56bu, 0x64b0398eu, 0x9ac3815eu, 0x14970614u, 0x6139a686u, 0x6a5eeda3u, 0xbb567051u, 0xd2cd3e4du, 0x08c46214u, 0xc15f2c97u
};
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
0xebd9055cu, 0x7eaf6b21u, 0x9b610855u, 0x134a8562u, 0xb10059bau, 0x7f459e58u, 0x8f3c7873u, 0x3523e609u,
0x75b8f4cau, 0x750d31b4u, 0x0dae0782u, 0x26f10015u, 0x7ba87a41u, 0x0efd8543u, 0x691d6368u, 0x8929d967u
};
static const uint32_t IGNEUM_CACHE_LAST[16] = {
0x51474edfu, 0xc3cfba93u, 0xf21454cfu, 0x9b79baacu, 0x4d4fce23u, 0xd701dfd5u, 0x37357ab3u, 0x1be693fau,
0xa701cb3bu, 0x7467c620u, 0x428184e8u, 0xf4010df0u, 0xe33aa88fu, 0xc78d6d62u, 0xae9ba9c0u, 0xf4bba97eu
};
static const uint64_t IGNEUM_CACHE_FNV64 = 0x448274a57f508cbcull;

View file

@ -0,0 +1,36 @@
{
"seed": "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000",
"day": "bytes:69676e65756d2d6461792ffa50000000000000",
"dataset_mode": "memory-hard",
"dataset_log2_words": 28,
"mask": "0x0fffffff",
"lanes": 32,
"source": "igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0xa32ad2dd06264089", "0x854fea26361763aa", "0xd958908d36ee63b5", "0x33641fe988c44ad9", "0xb5290b7892dbc7ba", "0xd815f329e6f806c0", "0x9fb28ff3fc29e39c", "0xdbc28a5d8bf42ab4",
"0x06f955808ff71bb9", "0x1ec10fff7bcdeadd", "0x1acbc289cddd83cd", "0x395e1564cda10137", "0x8a2aac2aa5b190d2", "0x499b8401e1f3e412", "0x9e2320319c85fc43", "0x66ba9661bc421ad6",
"0x3ba69971f1c9d8bd", "0x00d7a64d460bfefe", "0xc0f6487a1fbc5978", "0x1a261ac8673e1146", "0xf5a95a7caaaafa0a", "0x7bbdabd107842744", "0x49dd3bf9b77a16cf", "0xf7026cc441495a05",
"0x49fc48c1cba2be91", "0x84c2fc11de378cf9", "0x96de9e40c9da551c", "0x3d20f01e363aaf61", "0xef36922a5fcf96ca", "0x31f1b0c4b0b42aaf", "0x33c6406d516d5992", "0x847af3a4248a972f"
]},
{"base_nonce": 4096, "expected": [
"0x2039f40a61003341", "0x8238174df3761142", "0x32a72a04c4ca4a3c", "0x16aa0fd3f65da6b3", "0x502bf9db9bd98782", "0x109f60bc6167ff30", "0x3873c59792d6b552", "0xea98fba68fc32149",
"0x56a8440169cb0f83", "0x7362edc3a7c98252", "0x8f393a7ad9f4ff40", "0x6b3cb3a1e0b02453", "0xdcbdc40aedfa06c0", "0x2bc381227147a2f3", "0x1570ef95b0c8e31f", "0xf3ca64d0f9ca8c91",
"0x1c767a63cd6a0bf8", "0x8e1a13a1237a948b", "0xff2241b0b53b3efd", "0x95949c528978b1fb", "0xce3519499b78db7e", "0x56a0fcecaaa62096", "0x0a62c2bddc7041b2", "0x5ec463510d31d7c0",
"0x2fc74273ed47b17f", "0xedc5740b0c5c191e", "0xe62f737c106216cb", "0x185df496808afd70", "0x29070790e64a0bc7", "0xc0ddaeac8b22b14b", "0x302982837b6d96b4", "0x523972727aa906b5"
]},
{"base_nonce": 1000000, "expected": [
"0x3527292a5f4afb4e", "0x8c4128a03d946030", "0xb5b598dd18eb207d", "0x9f1b21b5bf9dce8f", "0xa791a53fa1a4733e", "0x12d2f9629aaca3b8", "0x400e844a00ba254c", "0x1e9f98bee0272859",
"0x9983064542aabb54", "0x017873e17fab719c", "0xf3e747e327968d2b", "0x562380f8016b9ca7", "0x3f5007c1ec21b4a8", "0x5612489676e34c2b", "0x97e13b52bc125dab", "0xe15cf6d8d12f9b00",
"0xd8ba508db7b12919", "0x2e16d66db0bd1837", "0x0026e228553d7b18", "0x0c6d22f5018958dd", "0xe76da566fc36580d", "0x7824813a77faff44", "0x663811c5504ff04d", "0x75efd5493353d65a",
"0x926b72ebe963caa2", "0x392723945438b76d", "0x4df61663eb2bfe05", "0x9a305821ba9ceb59", "0x317f4a6ac3f65a88", "0xdbae9b884f6829a9", "0x97cc8c8866b1a0cb", "0x5db142b6d7777df1"
]}
],
"dataset_head": ["0x3dd50b1f", "0x48edec90", "0x93123701", "0x3a7d2407", "0x5119540e", "0x657ac748", "0x3358ef2d", "0x09e2105c", "0xcdd88b45", "0xb03a0ac0", "0x2c37c6b0", "0x10e2c6a5", "0x25e2929c", "0x4317cabd", "0x5fa0cdcc", "0x6178ee52"],
"dataset_last_index": 268435455,
"dataset_last": "0xf7b7180e",
"dataset_samples": [{"index": 59471966, "value": "0x1faef64a"}, {"index": 217795994, "value": "0x068e4a54"}, {"index": 208353206, "value": "0x24551c62"}, {"index": 42483309, "value": "0x1234f539"}, {"index": 172547758, "value": "0xc15100f5"}, {"index": 148076330, "value": "0x2f53b6b0"}, {"index": 183853158, "value": "0xbab5fb76"}, {"index": 214389424, "value": "0x06edf897"}, {"index": 267488061, "value": "0xb0880fc6"}, {"index": 169781097, "value": "0xccfddc4d"}, {"index": 184093494, "value": "0xf640c0f4"}, {"index": 153880993, "value": "0xf145a054"}, {"index": 84977930, "value": "0xa2425e8f"}, {"index": 46426879, "value": "0x0a218118"}, {"index": 3093825, "value": "0xd310512a"}, {"index": 225364072, "value": "0xb612d7ff"}, {"index": 44593546, "value": "0x9a290b4d"}, {"index": 260713159, "value": "0xaf0227ed"}, {"index": 168250303, "value": "0x81b9c33c"}, {"index": 52384140, "value": "0x5f352dae"}, {"index": 223401610, "value": "0x46889c3d"}, {"index": 45554030, "value": "0x8ddc5d17"}, {"index": 95410555, "value": "0x368b4a6c"}, {"index": 175039924, "value": "0x16acb57a"}, {"index": 79171087, "value": "0x103f86c1"}, {"index": 267580473, "value": "0x7144efaf"}, {"index": 24168642, "value": "0x87cd0dc1"}, {"index": 37981670, "value": "0x1d9132b1"}, {"index": 171551130, "value": "0x28b0d87c"}, {"index": 195559979, "value": "0x01c44610"}, {"index": 204611762, "value": "0x908b6bac"}, {"index": 140997658, "value": "0x9e127820"}, {"index": 138925853, "value": "0x9e695a6f"}, {"index": 86637313, "value": "0xfbb80109"}, {"index": 20736778, "value": "0x39ad9ba2"}, {"index": 219665210, "value": "0x0c867f51"}, {"index": 160430336, "value": "0xac1e69e5"}, {"index": 264654675, "value": "0xf16f763d"}, {"index": 8013395, "value": "0x2de60ff0"}, {"index": 228945585, "value": "0xfabaf858"}, {"index": 213884386, "value": "0x53a35511"}, {"index": 104419827, "value": "0x249e753f"}, {"index": 44185464, "value": "0x88f84300"}, {"index": 142737231, "value": "0xd79db283"}, {"index": 99284897, "value": "0x7ac395ca"}, {"index": 132475900, "value": "0x5df23d8b"}, {"index": 61861762, "value": "0xe3f98318"}, {"index": 132056166, "value": "0x33ad1217"}, {"index": 262388043, "value": "0x2dd19e3f"}, {"index": 91878046, "value": "0xf1a97b54"}, {"index": 117353561, "value": "0xb1fabd1e"}, {"index": 124768597, "value": "0x1bc27b43"}, {"index": 71352993, "value": "0xb208c2e5"}, {"index": 190698941, "value": "0x1f109813"}, {"index": 46055428, "value": "0xccb6a56b"}, {"index": 55281366, "value": "0x64b0398e"}, {"index": 165145231, "value": "0x9ac3815e"}, {"index": 106810753, "value": "0x14970614"}, {"index": 171985651, "value": "0x6139a686"}, {"index": 232085256, "value": "0x6a5eeda3"}, {"index": 159510492, "value": "0xbb567051"}, {"index": 40072060, "value": "0xd2cd3e4d"}, {"index": 209107596, "value": "0x08c46214"}, {"index": 39023794, "value": "0xc15f2c97"}],
"cache_head": ["0xebd9055c", "0x7eaf6b21", "0x9b610855", "0x134a8562", "0xb10059ba", "0x7f459e58", "0x8f3c7873", "0x3523e609", "0x75b8f4ca", "0x750d31b4", "0x0dae0782", "0x26f10015", "0x7ba87a41", "0x0efd8543", "0x691d6368", "0x8929d967"],
"cache_last_line": ["0x51474edf", "0xc3cfba93", "0xf21454cf", "0x9b79baac", "0x4d4fce23", "0xd701dfd5", "0x37357ab3", "0x1be693fa", "0xa701cb3b", "0x7467c620", "0x428184e8", "0xf4010df0", "0xe33aa88f", "0xc78d6d62", "0xae9ba9c0", "0xf4bba97e"],
"cache_fnv1a64": "0x448274a57f508cbc"
}

View file

@ -0,0 +1,281 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 3 8 13 of w.
static inline uint mh_j(uint w) { return ((w >> 1u) & 1u) | (((w >> 3u) & 1u) << 1) | (((w >> 8u) & 1u) << 2) | (((w >> 13u) & 1u) << 3); }
static inline uint mh_t(uint w) { w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000000ffu) | ((w >> 9u) << 8u); w = (w & 0x00000007u) | ((w >> 4u) << 3u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
static inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 3u) << 4u) | (w & 0x00000007u) | (((j >> 1u) & 1u) << 3u); w = ((w >> 8u) << 9u) | (w & 0x000000ffu) | (((j >> 2u) & 1u) << 8u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 3u) & 1u) << 13u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
for (uint i = 0u; i < 16u; ++i) ds[(ulong)mh_addr(t, i)] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r6 = r6 ^ t_; } // 14 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r3 = r3 ^ t_; } // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = mul_hi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = mul_hi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r0 = r0 ^ t_; } // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = mul_hi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r5 = r5 ^ t_; } // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif

View file

@ -0,0 +1,163 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
#include "memhard.h"
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
if (t < nItems) {
uint32_t s[16];
mh_item(cache, t, s);
for (uint32_t i = 0u; i < 16u; ++i) ds[(size_t)mh_addr(t, i)] = s[i];
}
}
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint32_t x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint32_t x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint32_t x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint32_t x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint32_t x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint32_t x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint32_t x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 14 shfl
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = __umulhi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = __umulhi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = __umulhi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
if (nSegments == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nSegments + block - 1u) / block;
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
return cudaGetLastError();
}
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
if (nItems == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nItems + block - 1u) / block;
igneum_build<<<grid, block>>>(ds, cache, nItems);
return cudaGetLastError();
}
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps) {
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,375 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 3 8 13 of w.
static inline uint mh_j(uint w) { return ((w >> 1u) & 1u) | (((w >> 3u) & 1u) << 1) | (((w >> 8u) & 1u) << 2) | (((w >> 13u) & 1u) << 3); }
static inline uint mh_t(uint w) { w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000000ffu) | ((w >> 9u) << 8u); w = (w & 0x00000007u) | ((w >> 4u) << 3u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
static inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 3u) << 4u) | (w & 0x00000007u) | (((j >> 1u) & 1u) << 3u); w = ((w >> 8u) << 9u) | (w & 0x000000ffu) | (((j >> 2u) & 1u) << 8u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 3u) & 1u) << 13u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
for (uint i = 0u; i < 16u; ++i) ds[(ulong)mh_addr(t, i)] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x667d0fbdu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x7b8e5963u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x7b8e5963u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0x31c67e5eu; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0x31c67e5eu; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4529ddc6u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4529ddc6u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xef19d6d8u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xef19d6d8u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xaccf6211u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xaccf6211u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0xda0aed32u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0xda0aed32u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0xabc6df31u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0xabc6df31u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x667d0fbdu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r6 = r6 ^ t_; } // 14 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r3 = r3 ^ t_; } // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = mul_hi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = mul_hi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r0 = r0 ^ t_; } // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = mul_hi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r5 = r5 ^ t_; } // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {
uint gid = (uint)get_global_id(0);
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 1u); r3 = r3 ^ t_; } // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 16u); r6 = r6 ^ t_; } // 14 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 1u); r3 = r3 ^ t_; } // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = mul_hi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r7 = r7 ^ t_; } // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = mul_hi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
{ uint t_; IGNEUM_SHFL_XOR(t_, r7, 4u); r0 = r0 ^ t_; } // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = mul_hi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 8u); r5 = r5 ^ t_; } // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}

View file

@ -0,0 +1,123 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
// Host declarations (also in program_bound.h if present):
// struct IgneumInitWords { uint32_t w[8]; };
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
struct IgneumInitWords { uint32_t w[8]; };
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {
uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r4 = r4 + r5 + ((((sel >> 13u) & 1u) != 0u) ? 0x5810667au : 0xea86e152u); // 0 add
r7 = r7 ^ r0; // 1 xor
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r6, 1); // 2 shfl
r4 = r4 - r1; // 3 sub
r2 = r0 * r4 + r2; // 4 mad
r4 = rotl_imm(r4, 9u); // 5 rotl
r0 = rotr_var(r0, r2); // 6 rotr
r6 = r6 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x04000000u) & mask]; // 7 load
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 8 load
r1 = r1 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 9 load
r7 = r7 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 10 load
r7 = r7 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 11 load
r5 = r5 * r4; // 12 mul
r7 = r7 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 13 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r5, 16); // 14 shfl
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r5, 1); // 15 shfl
r2 = r2 | r7; // 16 or
r0 = rotr_var(r0, r6); // 17 rotr
r7 = r7 + r5 + ((((sel >> 12u) & 1u) != 0u) ? 0xb9e3577eu : 0xf66e7017u); // 18 add
r7 = rotl_imm(r7, 24u); // 19 rotl
r6 = r6 + r7 + ((((sel >> 23u) & 1u) != 0u) ? 0x8c9f0ef8u : 0x52334d12u); // 20 add
r2 = r2 | r6; // 21 or
r6 = r5 * r4 + r6; // 22 mad
r1 = r1 | r0; // 23 or
r6 = r6 ^ r1; // 24 xor
r2 = r2 + r6 + ((((sel >> 17u) & 1u) != 0u) ? 0x9cec0e12u : 0x659fc3d3u); // 25 add
r7 = rotr_var(r7, r0); // 26 rotr
r4 = r5 * r7 + r4; // 27 mad
r3 = r2 * r4 + r3; // 28 mad
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 29 load
r2 = r2 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x08000000u) & mask]; // 30 load
r1 = r1 ^ ds[((rotl_imm(r5 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 31 load
r7 = r7 + r2 + ((((sel >> 16u) & 1u) != 0u) ? 0xe403240eu : 0x070888a8u); // 32 add
r2 = r2 + r0 + ((((sel >> 28u) & 1u) != 0u) ? 0x29701828u : 0xf2e46d55u); // 33 add
r2 = r2 + r3 + ((((sel >> 14u) & 1u) != 0u) ? 0x343b7aeeu : 0x58f75b87u); // 34 add
r2 = __umulhi(r2, r5); // 35 mulhi
r4 = r4 ^ r2; // 36 xor
r6 = r6 * r5; // 37 mul
r7 = r7 ^ r0; // 38 xor
r7 = r7 + r2 + ((((sel >> 19u) & 1u) != 0u) ? 0x32bbd117u : 0xb8180e9du); // 39 add
r2 = rotr_var(r2, r3); // 40 rotr
r7 = r7 - r0; // 41 sub
r4 = r4 + r3 + ((((sel >> 4u) & 1u) != 0u) ? 0x6df7aed4u : 0x6ced15b7u); // 42 add
r7 = r7 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 43 shfl
r0 = r0 ^ ds[((rotl_imm(r7 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 44 load
r1 = r1 + r6 + ((((sel >> 14u) & 1u) != 0u) ? 0x83e825bfu : 0xe09f54e9u); // 45 add
r3 = r3 ^ ds[((rotl_imm(r1 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 46 load
r6 = r6 ^ ds[((rotl_imm(r3 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 47 load
r4 = __umulhi(r4, r2); // 48 mulhi
r5 = r5 + r0 + ((((sel >> 7u) & 1u) != 0u) ? 0xf572bdb9u : 0xa8bae6dfu); // 49 add
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r7, 4); // 50 shfl
r6 = r6 + r0 + ((((sel >> 19u) & 1u) != 0u) ? 0x11e17c61u : 0x383b9260u); // 51 add
r5 = r5 ^ ds[((rotl_imm(r2 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 52 load
r6 = rotl_imm(r6, 12u); // 53 rotl
r3 = rotl_imm(r3, 12u); // 54 rotl
r2 = r2 + r1 + ((((sel >> 10u) & 1u) != 0u) ? 0x18d67dbbu : 0xac6be8e3u); // 55 add
r1 = r1 ^ ds[((rotl_imm(r4 * 0xb2a9d70du, 6u) & 0x0fffffffu) | 0x00000000u) & mask]; // 56 load
r5 = r5 + r0 + ((((sel >> 12u) & 1u) != 0u) ? 0xa75cd60du : 0xf03673feu); // 57 add
r5 = r5 ^ ds[((rotl_imm(r0 * 0xb2a9d70du, 6u) & 0x03ffffffu) | 0x00000000u) & mask]; // 58 load
r1 = r1 + r4 + ((((sel >> 7u) & 1u) != 0u) ? 0x8dfb96bbu : 0xdecd4794u); // 59 add
r3 = __umulhi(r3, r2); // 60 mulhi
r6 = r6 | r4; // 61 or
r5 = r5 ^ __shfl_xor_sync(0xffffffffu, r4, 8); // 62 shfl
r3 = r3 ^ ds[((rotl_imm(r6 * 0xb2a9d70du, 6u) & 0x07ffffffu) | 0x08000000u) & mask]; // 63 load
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {
if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;
igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);
return cudaGetLastError();
}
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,112 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#elif defined(_MSC_VER) && !defined(__cplusplus)
#define IGNEUM_HD static __inline
#else
#define IGNEUM_HD static inline
#endif
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint32_t r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint32_t r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 3 8 13 of w.
IGNEUM_HD uint32_t mh_j(uint32_t w) { return ((w >> 1u) & 1u) | (((w >> 3u) & 1u) << 1) | (((w >> 8u) & 1u) << 2) | (((w >> 13u) & 1u) << 3); }
IGNEUM_HD uint32_t mh_t(uint32_t w) { w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000000ffu) | ((w >> 9u) << 8u); w = (w & 0x00000007u) | ((w >> 4u) << 3u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
IGNEUM_HD uint32_t mh_addr(uint32_t t, uint32_t j) { uint32_t w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 3u) << 4u) | (w & 0x00000007u) | (((j >> 1u) & 1u) << 3u); w = ((w >> 8u) << 9u) | (w & 0x000000ffu) | (((j >> 2u) & 1u) << 8u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 3u) & 1u) << 13u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }

View file

@ -0,0 +1,109 @@
#include <metal_stdlib>
using namespace metal;
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
inline void mh_cache_segment(device uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0xceed56d7u ^ prev[4];
x[5] = 0x9ba270d2u ^ prev[5];
x[6] = 0x82caab2du ^ prev[6];
x[7] = 0x81ebce0eu ^ prev[7];
x[8] = 0x12b6ecf1u ^ prev[8];
x[9] = 0xd0f3fd7cu ^ prev[9];
x[10] = 0xd872eefeu ^ prev[10];
x[11] = 0xc158c7bdu ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
inline void mh_mixer(thread uint* s, uint rk) {
s[0] = (s[0] ^ (0xc6892460u + rk)) * 0xf351d601u;
s[1] = (s[1] ^ (0x25b7228au + rk)) * 0xa3bb398fu;
s[2] = (s[2] ^ (0xcd515004u + rk)) * 0xb5a09e35u;
s[3] = (s[3] ^ (0x2846527au + rk)) * 0x7509c9c1u;
s[4] = (s[4] ^ (0xa6324241u + rk)) * 0x6bbf31e9u;
s[5] = (s[5] ^ (0x36e3ec53u + rk)) * 0xfc849a79u;
s[6] = (s[6] ^ (0x82961bacu + rk)) * 0xded91851u;
s[7] = (s[7] ^ (0x0f97ba7du + rk)) * 0x8d9113d1u;
s[8] = (s[8] ^ (0xb6f921a9u + rk)) * 0x0ff15225u;
s[9] = (s[9] ^ (0x3ada24e5u + rk)) * 0x3a5bdd41u;
s[10] = (s[10] ^ (0xde20ab91u + rk)) * 0xab533435u;
s[11] = (s[11] ^ (0x5378eeb2u + rk)) * 0xe1c55ad5u;
s[12] = (s[12] ^ (0x7d161662u + rk)) * 0xe6d3bd0du;
s[13] = (s[13] ^ (0x89353cc1u + rk)) * 0x9d9ffbbdu;
s[14] = (s[14] ^ (0xb1aa03a2u + rk)) * 0xbb2a3cf3u;
s[15] = (s[15] ^ (0x788acae6u + rk)) * 0x50a7c08du;
MH_QR(s[0], s[4], s[8], s[12], 17u, 12u, 20u, 23u) MH_QR(s[1], s[5], s[9], s[13], 17u, 12u, 20u, 23u)
MH_QR(s[2], s[6], s[10], s[14], 17u, 12u, 20u, 23u) MH_QR(s[3], s[7], s[11], s[15], 17u, 12u, 20u, 23u)
MH_QR(s[0], s[5], s[10], s[15], 7u, 3u, 27u, 16u) MH_QR(s[1], s[6], s[11], s[12], 7u, 3u, 27u, 16u)
MH_QR(s[2], s[7], s[8], s[13], 7u, 3u, 27u, 16u) MH_QR(s[3], s[4], s[9], s[14], 7u, 3u, 27u, 16u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
s[0] = 0xceed56d7u;
s[1] = 0x9ba270d2u;
s[2] = 0x82caab2du;
s[3] = 0x81ebce0eu;
s[4] = 0x12b6ecf1u;
s[5] = 0xd0f3fd7cu;
s[6] = 0xd872eefeu;
s[7] = 0xc158c7bdu;
s[8] = t * 0xf351d601u + 0xc6892460u;
s[9] = t * 0xa3bb398fu + 0x25b7228au;
s[10] = t * 0xb5a09e35u + 0xcd515004u;
s[11] = t * 0x7509c9c1u + 0x2846527au;
s[12] = t * 0x6bbf31e9u + 0xa6324241u;
s[13] = t * 0xfc849a79u + 0x36e3ec53u;
s[14] = t * 0xded91851u + 0x82961bacu;
s[15] = t * 0x8d9113d1u + 0x0f97ba7du;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions 1 3 8 13 of w.
inline uint mh_j(uint w) { return ((w >> 1u) & 1u) | (((w >> 3u) & 1u) << 1) | (((w >> 8u) & 1u) << 2) | (((w >> 13u) & 1u) << 3); }
inline uint mh_t(uint w) { w = (w & 0x00001fffu) | ((w >> 14u) << 13u); w = (w & 0x000000ffu) | ((w >> 9u) << 8u); w = (w & 0x00000007u) | ((w >> 4u) << 3u); w = (w & 0x00000001u) | ((w >> 2u) << 1u); return w; }
inline uint mh_addr(uint t, uint j) { uint w = t; w = ((w >> 1u) << 2u) | (w & 0x00000001u) | (((j >> 0u) & 1u) << 1u); w = ((w >> 3u) << 4u) | (w & 0x00000007u) | (((j >> 1u) & 1u) << 3u); w = ((w >> 8u) << 9u) | (w & 0x000000ffu) | (((j >> 2u) & 1u) << 8u); w = ((w >> 13u) << 14u) | (w & 0x00001fffu) | (((j >> 3u) & 1u) << 13u); return w; }
// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }
// One thread per segment (2^16 threads).
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
mh_cache_segment(cache, gid);
}
// One thread per 64-byte item (dataset words / 16 threads).
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
uint s[16];
mh_item(cache, gid, s);
for (uint i = 0u; i < 16u; ++i) dataset[mh_addr(gid, i)] = s[i];
}

View file

@ -0,0 +1,76 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#ifndef IGNEUM_NO_CUDA
#include <cuda_runtime.h>
#endif
#define IGNEUM_SEED_STRING "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000"
#define IGNEUM_SEED_BYTES_HEX "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"
#define IGNEUM_GENERATOR 3
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0x73bcbfe8ccf988f1ull
#define IGNEUM_DAY_STRING "bytes:69676e65756d2d6461792ffa50000000000000"
#define IGNEUM_DAY_BYTES_HEX "69676e65756d2d6461792ffa50000000000000"
#define IGNEUM_DAY0 0xceed56d7u
#define IGNEUM_DAY1 0x9ba270d2u
#define IGNEUM_DATASET_LOG2 28
#define IGNEUM_MASK 0x0fffffffu
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS 8
#define IGNEUM_INSTR_COUNT 64
#define IGNEUM_LOADS_PER_HASH 128
#define IGNEUM_WIDE_LOADS_PER_HASH 0
#define IGNEUM_OP_MIX "load=16 add=15 shfl=6 mad=4 or=4 rotl=4 rotr=4 xor=4 mulhi=3 mul=2 sub=2"
// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that
// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=<hex>).
#define IGNEUM_PROGRAM_CLASS "v3"
#define IGNEUM_ERA_SEED_HEX "ff87ad96a1b53f367a95d5ca123bab64211bd9aa57fc7ad3c5e2d496c20138d4"
// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
#define IGNEUM_LOAD_CLASS "w4-era676a17fc"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 512
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Era layout (5 October 2026, docs/plans/era-layout.md): NOT the lottery hash. Every dataset load reads
// idx = ((rotl(src * STRIDE_MUL, STRIDE_ROT) & window mask) | window offset) & MASK; the window of a load site is the
// dataset, a half or a quarter of it (IGNEUM_ERA_WINDOWS: site:shrink:offset); dataset word w holds word j(w) of item
// t(w) with j's bits at the INTERLEAVE positions (memhard.h: mh_t, mh_j, mh_addr).
#define IGNEUM_ERA_LABEL "676a17fc"
#define IGNEUM_ERA_SEED_WORDS { 0x676a17fcu, 0x60bc956eu, 0x3e9865f4u, 0x68ae9e61u, 0xc0b3f442u, 0xaf406bebu, 0xb6126b9au, 0xaa30959bu }
#define IGNEUM_ERA_ALLOWED_WIDTHS { 1, 0, 0 } // words, ascending, 0 = unused; one entry pins the width
#define IGNEUM_ERA_WIDTH_WORDS 1
#define IGNEUM_ERA_STRIDE_MUL 0xb2a9d70du
#define IGNEUM_ERA_STRIDE_ROT 6
#define IGNEUM_ERA_INTERLEAVE { 1, 3, 8, 13 }
#define IGNEUM_ERA_WINDOWS "7:2:1 8:1:1 9:1:1 10:1:1 11:0:0 13:1:1 29:0:0 30:2:2 31:1:1 44:1:1 46:2:0 47:0:0 52:0:0 56:0:0 58:2:0 63:1:1"
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1
#define IGNEUM_SEEDW_INIT { 0x667d0fbdu, 0x7b8e5963u, 0x31c67e5eu, 0x4529ddc6u, 0xef19d6d8u, 0xaccf6211u, 0xda0aed32u, 0xabc6df31u }
#define IGNEUM_KEY_INIT { 0xceed56d7u, 0x9ba270d2u, 0x82caab2du, 0x81ebce0eu, 0x12b6ecf1u, 0xd0f3fd7cu, 0xd872eefeu, 0xc158c7bdu }
#define IGNEUM_CACHE_LOG2_WORDS 26
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
#define IGNEUM_CACHE_SEGMENTS 65536u
#define IGNEUM_ITEM_ROUNDS 8
#define IGNEUM_MIX_ROT_INIT { 17u, 12u, 20u, 23u, 7u, 3u, 27u, 16u }
#define IGNEUM_MIX_MUL_INIT { 0xf351d601u, 0xa3bb398fu, 0xb5a09e35u, 0x7509c9c1u, 0x6bbf31e9u, 0xfc849a79u, 0xded91851u, 0x8d9113d1u, 0x0ff15225u, 0x3a5bdd41u, 0xab533435u, 0xe1c55ad5u, 0xe6d3bd0du, 0x9d9ffbbdu, 0xbb2a3cf3u, 0x50a7c08du }
#define IGNEUM_MIX_RC_INIT { 0xc6892460u, 0x25b7228au, 0xcd515004u, 0x2846527au, 0xa6324241u, 0x36e3ec53u, 0x82961bacu, 0x0f97ba7du, 0xb6f921a9u, 0x3ada24e5u, 0xde20ab91u, 0x5378eeb2u, 0x7d161662u, 0x89353cc1u, 0xb1aa03a2u, 0x788acae6u }
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#endif

View file

@ -0,0 +1,143 @@
{
"format": "igneum-program-pack-3",
"generator": 3,
"attempt": 0,
"program_id": "0x73bcbfe8ccf988f1",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07/day/69676e65756d2d6461792ffa50000000000000",
"seed_bytes": "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07",
"seed_words": ["0x667d0fbd", "0x7b8e5963", "0x31c67e5e", "0x4529ddc6", "0xef19d6d8", "0xaccf6211", "0xda0aed32", "0xabc6df31"],
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
"lanes": 32,
"registers": 8,
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"program_class": "v3",
"era_seed_bytes": "ff87ad96a1b53f367a95d5ca123bab64211bd9aa57fc7ad3c5e2d496c20138d4",
"load_class": "w4-era676a17fc",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [16, 0, 0],
"bytes_per_hash": 512,
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"era": {
"label": "676a17fc",
"seed_words": ["0x676a17fc", "0x60bc956e", "0x3e9865f4", "0x68ae9e61", "0xc0b3f442", "0xaf406beb", "0xb6126b9a", "0xaa30959b"],
"draw": "docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used",
"allowed_widths": [1],
"width_words": 1,
"stride_mul": "0xb2a9d70d",
"stride_rot": 6,
"interleave": [1, 3, 8, 13],
"address": "y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words",
"windows": "per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)",
"dataset_word": "dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed",
"program_id_suffix": "'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]"
},
"op_mix": {"load": 16, "add": 15, "shfl": 6, "mad": 4, "or": 4, "rotl": 4, "rotr": 4, "xor": 4, "mulhi": 3, "mul": 2, "sub": 2},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
"op_semantics": {
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
"sub": "dst = dst - src",
"mul": "dst = dst * src (low 32)",
"mulhi": "dst = high 32 bits of dst * src",
"xor": "dst = dst ^ src",
"or": "dst = dst | src",
"rotl": "dst = rotl(dst, rot), rot in 1..31",
"rotr": "dst = rotr(dst, src & 31)",
"mad": "dst = src * src2 + dst",
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
"load": "dst = dst ^ dataset[src & dataset.mask]",
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
},
"dataset": {
"log2_words": 28,
"bytes": 1073741824,
"mask": "0x0fffffff",
"day": "bytes:69676e65756d2d6461792ffa50000000000000",
"day_bytes": "69676e65756d2d6461792ffa50000000000000",
"day_words_from": "seed_words_from_bytes(day_bytes)",
"d0": "0xceed56d7",
"d1": "0x9ba270d2",
"mode": "memory-hard",
"spec": "proto-metal/MEMHARD.md",
"key": ["0xceed56d7", "0x9ba270d2", "0x82caab2d", "0x81ebce0e", "0x12b6ecf1", "0xd0f3fd7c", "0xd872eefe", "0xc158c7bd"],
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [17, 12, 20, 23, 7, 3, 27, 16], "mul": ["0xf351d601", "0xa3bb398f", "0xb5a09e35", "0x7509c9c1", "0x6bbf31e9", "0xfc849a79", "0xded91851", "0x8d9113d1", "0x0ff15225", "0x3a5bdd41", "0xab533435", "0xe1c55ad5", "0xe6d3bd0d", "0x9d9ffbbd", "0xbb2a3cf3", "0x50a7c08d"], "rc": ["0xc6892460", "0x25b7228a", "0xcd515004", "0x2846527a", "0xa6324241", "0x36e3ec53", "0x82961bac", "0x0f97ba7d", "0xb6f921a9", "0x3ada24e5", "0xde20ab91", "0x5378eeb2", "0x7d161662", "0x89353cc1", "0xb1aa03a2", "0x788acae6"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
"word": "dataset[w] = item(w >> 4)[w & 15]"
},
"instructions": [
{"i": 0, "op": "add", "dst": 4, "src": 5, "src2": 7, "imm": "0xea86e152", "imm2": "0x5810667a", "rot": 27, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 1, "op": "xor", "dst": 7, "src": 0, "src2": 6, "imm": "0xbaab6229", "imm2": "0xed861989", "rot": 26, "bit": 22, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 2, "op": "shfl", "dst": 3, "src": 6, "src2": 2, "imm": "0x5b623116", "imm2": "0xff12e5b2", "rot": 12, "bit": 24, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 3, "op": "sub", "dst": 4, "src": 1, "src2": 1, "imm": "0xe99741c7", "imm2": "0xf5fa5009", "rot": 1, "bit": 21, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 4, "op": "mad", "dst": 2, "src": 0, "src2": 4, "imm": "0x673c2157", "imm2": "0xee02465f", "rot": 20, "bit": 22, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 5, "op": "rotl", "dst": 4, "src": 7, "src2": 7, "imm": "0x946f7818", "imm2": "0x45d3399e", "rot": 9, "bit": 2, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 6, "op": "rotr", "dst": 0, "src": 2, "src2": 4, "imm": "0x5f6a0ed2", "imm2": "0x7043a636", "rot": 19, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 7, "op": "load", "dst": 6, "src": 7, "src2": 1, "imm": "0x5c61dcf7", "imm2": "0x7466aa40", "rot": 19, "bit": 9, "mask": 2, "width": 1, "win": 2, "off": 1},
{"i": 8, "op": "load", "dst": 1, "src": 4, "src2": 5, "imm": "0x85668475", "imm2": "0xdb8cc483", "rot": 29, "bit": 7, "mask": 4, "width": 1, "win": 1, "off": 1},
{"i": 9, "op": "load", "dst": 1, "src": 2, "src2": 4, "imm": "0x4454980f", "imm2": "0xebd31581", "rot": 10, "bit": 28, "mask": 16, "width": 1, "win": 1, "off": 1},
{"i": 10, "op": "load", "dst": 7, "src": 0, "src2": 2, "imm": "0xe075297c", "imm2": "0x5779c44c", "rot": 10, "bit": 22, "mask": 2, "width": 1, "win": 1, "off": 1},
{"i": 11, "op": "load", "dst": 7, "src": 1, "src2": 6, "imm": "0x65aa4311", "imm2": "0x4fe48ea9", "rot": 15, "bit": 9, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 12, "op": "mul", "dst": 5, "src": 4, "src2": 2, "imm": "0x1383d3ad", "imm2": "0xf3094b29", "rot": 8, "bit": 9, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 13, "op": "load", "dst": 7, "src": 6, "src2": 2, "imm": "0xed8a496f", "imm2": "0x3072c3c6", "rot": 28, "bit": 19, "mask": 8, "width": 1, "win": 1, "off": 1},
{"i": 14, "op": "shfl", "dst": 6, "src": 5, "src2": 0, "imm": "0x8b965b57", "imm2": "0xcfeca6c1", "rot": 12, "bit": 27, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 15, "op": "shfl", "dst": 3, "src": 5, "src2": 3, "imm": "0x877c7586", "imm2": "0xa9cb2a03", "rot": 2, "bit": 29, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 16, "op": "or", "dst": 2, "src": 7, "src2": 3, "imm": "0xb740221a", "imm2": "0x89d38d6d", "rot": 6, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 17, "op": "rotr", "dst": 0, "src": 6, "src2": 1, "imm": "0x26f3ad8a", "imm2": "0x27256f15", "rot": 18, "bit": 5, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 18, "op": "add", "dst": 7, "src": 5, "src2": 7, "imm": "0xf66e7017", "imm2": "0xb9e3577e", "rot": 9, "bit": 12, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 19, "op": "rotl", "dst": 7, "src": 0, "src2": 5, "imm": "0x849ae6ee", "imm2": "0x02b358f9", "rot": 24, "bit": 18, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 20, "op": "add", "dst": 6, "src": 7, "src2": 6, "imm": "0x52334d12", "imm2": "0x8c9f0ef8", "rot": 11, "bit": 23, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 21, "op": "or", "dst": 2, "src": 6, "src2": 5, "imm": "0xb1871e63", "imm2": "0xb2e40191", "rot": 5, "bit": 13, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 22, "op": "mad", "dst": 6, "src": 5, "src2": 4, "imm": "0x97df29e4", "imm2": "0xe60fea84", "rot": 11, "bit": 16, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 23, "op": "or", "dst": 1, "src": 0, "src2": 3, "imm": "0x8f30d21d", "imm2": "0x2df685a0", "rot": 31, "bit": 0, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 24, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0xd2c4025f", "imm2": "0x5269eb4d", "rot": 31, "bit": 21, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 25, "op": "add", "dst": 2, "src": 6, "src2": 5, "imm": "0x659fc3d3", "imm2": "0x9cec0e12", "rot": 6, "bit": 17, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 26, "op": "rotr", "dst": 7, "src": 0, "src2": 1, "imm": "0xd89ef484", "imm2": "0x20be3846", "rot": 12, "bit": 9, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 27, "op": "mad", "dst": 4, "src": 5, "src2": 7, "imm": "0xb48420ae", "imm2": "0x3d1f2485", "rot": 14, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 28, "op": "mad", "dst": 3, "src": 2, "src2": 4, "imm": "0xc5c46d76", "imm2": "0x700044b5", "rot": 22, "bit": 10, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 29, "op": "load", "dst": 1, "src": 4, "src2": 3, "imm": "0x0fbaf177", "imm2": "0xfff4f2ed", "rot": 20, "bit": 31, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 30, "op": "load", "dst": 2, "src": 3, "src2": 0, "imm": "0xe90eb2e5", "imm2": "0xb0f9eb79", "rot": 27, "bit": 14, "mask": 4, "width": 1, "win": 2, "off": 2},
{"i": 31, "op": "load", "dst": 1, "src": 5, "src2": 2, "imm": "0x97ba3fc3", "imm2": "0x7894e657", "rot": 3, "bit": 30, "mask": 16, "width": 1, "win": 1, "off": 1},
{"i": 32, "op": "add", "dst": 7, "src": 2, "src2": 7, "imm": "0x070888a8", "imm2": "0xe403240e", "rot": 2, "bit": 16, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 33, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0xf2e46d55", "imm2": "0x29701828", "rot": 31, "bit": 28, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 34, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x58f75b87", "imm2": "0x343b7aee", "rot": 12, "bit": 14, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 35, "op": "mulhi", "dst": 2, "src": 5, "src2": 6, "imm": "0x5892a9e6", "imm2": "0xc9824c94", "rot": 19, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 36, "op": "xor", "dst": 4, "src": 2, "src2": 3, "imm": "0x7b5b5474", "imm2": "0x45cfc5dd", "rot": 17, "bit": 18, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 37, "op": "mul", "dst": 6, "src": 5, "src2": 7, "imm": "0xccf564a5", "imm2": "0x873ad101", "rot": 7, "bit": 11, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 38, "op": "xor", "dst": 7, "src": 0, "src2": 6, "imm": "0xcac8f06d", "imm2": "0x6b97c683", "rot": 18, "bit": 28, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 39, "op": "add", "dst": 7, "src": 2, "src2": 4, "imm": "0xb8180e9d", "imm2": "0x32bbd117", "rot": 23, "bit": 19, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 40, "op": "rotr", "dst": 2, "src": 3, "src2": 0, "imm": "0x2d6070bc", "imm2": "0x68ff101e", "rot": 13, "bit": 18, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 41, "op": "sub", "dst": 7, "src": 0, "src2": 2, "imm": "0x0daf96ea", "imm2": "0x36f37be1", "rot": 5, "bit": 0, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 42, "op": "add", "dst": 4, "src": 3, "src2": 6, "imm": "0x6ced15b7", "imm2": "0x6df7aed4", "rot": 19, "bit": 4, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 43, "op": "shfl", "dst": 7, "src": 3, "src2": 5, "imm": "0x8ace05f3", "imm2": "0xd378ec12", "rot": 23, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 44, "op": "load", "dst": 0, "src": 7, "src2": 4, "imm": "0xb0607786", "imm2": "0xc4acabbc", "rot": 13, "bit": 7, "mask": 16, "width": 1, "win": 1, "off": 1},
{"i": 45, "op": "add", "dst": 1, "src": 6, "src2": 5, "imm": "0xe09f54e9", "imm2": "0x83e825bf", "rot": 23, "bit": 14, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 46, "op": "load", "dst": 3, "src": 1, "src2": 0, "imm": "0x63cc1e4e", "imm2": "0xa1be8118", "rot": 12, "bit": 6, "mask": 2, "width": 1, "win": 2, "off": 0},
{"i": 47, "op": "load", "dst": 6, "src": 3, "src2": 7, "imm": "0x353f1d79", "imm2": "0x3b2e7456", "rot": 18, "bit": 18, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 48, "op": "mulhi", "dst": 4, "src": 2, "src2": 7, "imm": "0x00d8a3cd", "imm2": "0x231866d2", "rot": 21, "bit": 20, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 49, "op": "add", "dst": 5, "src": 0, "src2": 2, "imm": "0xa8bae6df", "imm2": "0xf572bdb9", "rot": 14, "bit": 7, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 50, "op": "shfl", "dst": 0, "src": 7, "src2": 7, "imm": "0x81ef22e1", "imm2": "0x74438fc5", "rot": 28, "bit": 18, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 51, "op": "add", "dst": 6, "src": 0, "src2": 6, "imm": "0x383b9260", "imm2": "0x11e17c61", "rot": 12, "bit": 19, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 52, "op": "load", "dst": 5, "src": 2, "src2": 2, "imm": "0xfb84f451", "imm2": "0x11cd863e", "rot": 21, "bit": 20, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 53, "op": "rotl", "dst": 6, "src": 5, "src2": 4, "imm": "0xb1a7db6b", "imm2": "0x76686b9b", "rot": 12, "bit": 4, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 54, "op": "rotl", "dst": 3, "src": 6, "src2": 3, "imm": "0x6f981f52", "imm2": "0xd99aeba2", "rot": 12, "bit": 27, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 55, "op": "add", "dst": 2, "src": 1, "src2": 2, "imm": "0xac6be8e3", "imm2": "0x18d67dbb", "rot": 26, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 56, "op": "load", "dst": 1, "src": 4, "src2": 0, "imm": "0x7e7f6a00", "imm2": "0x6f0747da", "rot": 25, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 57, "op": "add", "dst": 5, "src": 0, "src2": 4, "imm": "0xf03673fe", "imm2": "0xa75cd60d", "rot": 16, "bit": 12, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 58, "op": "load", "dst": 5, "src": 0, "src2": 2, "imm": "0x227f94a6", "imm2": "0x0e8344f9", "rot": 20, "bit": 10, "mask": 2, "width": 1, "win": 2, "off": 0},
{"i": 59, "op": "add", "dst": 1, "src": 4, "src2": 3, "imm": "0xdecd4794", "imm2": "0x8dfb96bb", "rot": 21, "bit": 7, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 60, "op": "mulhi", "dst": 3, "src": 2, "src2": 2, "imm": "0x0dd268e0", "imm2": "0x53034ca9", "rot": 1, "bit": 8, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 61, "op": "or", "dst": 6, "src": 4, "src2": 7, "imm": "0x3a45a321", "imm2": "0x9bc59a5f", "rot": 25, "bit": 11, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 62, "op": "shfl", "dst": 5, "src": 4, "src2": 2, "imm": "0x8f229cc1", "imm2": "0xcaac64a2", "rot": 17, "bit": 13, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 63, "op": "load", "dst": 3, "src": 6, "src2": 6, "imm": "0xb2574178", "imm2": "0xcbcc798d", "rot": 28, "bit": 0, "mask": 16, "width": 1, "win": 1, "off": 1}
]
}

Some files were not shown because too many files have changed in this diff Show more