Merge branch 'master' into prover-floor

# Conflicts:
#	docs/bench-log.md
#	tools/prover-floor/pc2-floor-measure.ps1
#	tools/prover-floor/pc2-floor-sweep3.ps1
This commit is contained in:
igneum-labs 2026-10-06 12:22:22 +00:00
commit 7e330064bb
732 changed files with 110720 additions and 1490 deletions

View file

@ -25,6 +25,8 @@ jobs:
- name: igneum-pow tests (release)
working-directory: igneum-pow
run: cargo test --release
- name: pack loader seed rule (packfile.h on a known-good and a known-mismatched pack)
run: bash proto-cuda/nvrtc/emu/packfile-test.sh
- name: igneum-census build (release)
working-directory: igneum-census
run: cargo build --release
@ -67,8 +69,20 @@ jobs:
run: bash tools/ci/no-conflict-markers.sh
- name: copied sources are re-stamped before a build
run: bash tools/ci/copied-sources-check.sh
- name: second-engine playbooks log to a file and end their tree (C35)
run: bash tools/ci/second-engine-check.sh
- name: the signer is never piped into head
run: bash tools/ci/signer-pipe-check.sh
- name: bash bodies in PowerShell job scripts pass bash -n, the lost-quote class (self-test first, then the tree)
run: bash tools/ci/bash-body-check.sh --self-test && bash tools/ci/bash-body-check.sh
- name: run jobs test their fetched kit before use, the wiped-jobs-folder class (self-test first, then the tree)
run: bash tools/ci/kit-path-check.sh --self-test && bash tools/ci/kit-path-check.sh
- name: every Windows spawn of the app runs with a hidden console (self-test first, then the tree)
run: node tools/ci/windows-spawn-check.mjs --self-test && node tools/ci/windows-spawn-check.mjs
- name: pinned guest programs match their manifest and are built only by pin-guests.sh
run: bash tools/ci/pinned-guests-check.sh
- name: root prover playbooks kill the GPU server and unlink its socket (the root-socket class, 5 October 2026)
run: bash tools/ci/prover-socket-check.sh
- name: no secret file names and no 64-hex secrets in the tree (self-test first, then the tree)
run: bash tools/ci/no-secrets-check.sh --self-test && bash tools/ci/no-secrets-check.sh
- name: faucet unit tests (validation, the daily limits, the signed transaction; keccak, RLP and secp256k1 vectors)
@ -81,6 +95,6 @@ jobs:
- name: ship tool self-test (version bump, the dl-both and public manifest helpers)
run: node tools/ship-app.mjs --self-test
- name: relay unit tests (parsers, secret compare, the wake endpoint)
run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs
run: node --test relay/test/parse.test.mjs relay/test/auth.test.mjs relay/test/wake.test.mjs relay/test/ember.test.mjs
- name: miner app notice strip and update card (ordering, keys, wording, timers, when the card shows)
run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs
run: node --test app/igneum-app/ui/notices.test.mjs app/igneum-app/ui/update-card.test.mjs app/igneum-app/ui/view.test.mjs app/igneum-app/ui/tune-line.test.mjs

View file

@ -219,7 +219,7 @@ dependencies = [
[[package]]
name = "igneum-app"
version = "0.3.9"
version = "0.3.12"
dependencies = [
"ed25519-dalek",
"getrandom",

View file

@ -1,6 +1,6 @@
[package]
name = "igneum-app"
version = "0.3.9"
version = "0.3.12"
edition = "2021"
description = "Igneum Miner engine: supervises the node, the miner and the GPU workers, and serves the dashboard on 127.0.0.1"
license = "MIT"

View file

@ -6,8 +6,8 @@
1 ICON "igneum.ico"
1 VERSIONINFO
FILEVERSION 0,3,9,0
PRODUCTVERSION 0,3,9,0
FILEVERSION 0,3,12,0
PRODUCTVERSION 0,3,12,0
FILEFLAGSMASK 0x3fL
FILEFLAGS 0x0L
FILEOS VOS_NT_WINDOWS32
@ -20,12 +20,12 @@ BEGIN
BEGIN
VALUE "CompanyName", "Igneum"
VALUE "FileDescription", "Igneum Miner engine"
VALUE "FileVersion", "0.3.9"
VALUE "FileVersion", "0.3.12"
VALUE "InternalName", "igneum-app"
VALUE "LegalCopyright", "Igneum contributors"
VALUE "OriginalFilename", "igneum-app.exe"
VALUE "ProductName", "Igneum Miner"
VALUE "ProductVersion", "0.3.9"
VALUE "ProductVersion", "0.3.12"
END
END
BLOCK "VarFileInfo"

View file

@ -29,6 +29,16 @@ pub struct CardPref {
pub sweep_watts: f64,
#[serde(default)]
pub sweep_mhs: f64,
/// Ember Tune (src/ember.rs): the clock cap the last tune chose (0 = unlocked), the driver and program class it
/// ran under (a change makes the card due again), and the plan that produced it (full | confirm | baseline)
#[serde(default)]
pub sweep_clock_mhz: u32,
#[serde(default)]
pub sweep_driver: String,
#[serde(default)]
pub sweep_class: String,
#[serde(default)]
pub sweep_source: String,
}
#[derive(Clone, Serialize, Deserialize)]
@ -70,9 +80,16 @@ pub struct Settings {
#[serde(default)]
pub prove: bool,
/// The efficiency sweep (src/sweep.rs): once after install, then weekly, each NVIDIA card's cap is stepped from
/// 100% to 50% on the live program and held at the best MH per watt. Default on. A pinned card is skipped.
#[serde(default = "yes")]
/// 100% to 50% on the live program and held at the best MH per watt. Default off; implied by `power_control`
/// (on when that is switched on, never effective while it is off). A pinned card is skipped.
#[serde(default)]
pub sweep: bool,
/// Power control (the project lead, 5 October 2026: "if we don't have to ask then don't ask"): the NVIDIA power cap and the
/// efficiency sweep need administrator rights (one UAC prompt on Windows). Default OFF on every machine; the app
/// never raises the prompt on its own. Switching it on asks once, at that moment; a refused, cancelled or
/// unanswered prompt switches it back off with a notice, no retries.
#[serde(default)]
pub power_control: bool,
/// When this install first ran (unix s), for the "first hour after install" sweep.
#[serde(default)]
pub installed_at: u64,
@ -87,6 +104,10 @@ pub struct Settings {
/// so it includes proof records it never verified (src/verifier.rs). Default off; a found verifier always wins.
#[serde(default)]
pub proof_verify_trust: bool,
/// Proving v1 step 1 (5 October 2026): the install-time default for `prove` has been applied once (src/provedefault.rs:
/// on when the machine can prove, never switching an explicit on back off). Older installs apply it at their next start.
#[serde(default)]
pub prove_default_applied: bool,
}
fn one() -> u32 {
@ -98,16 +119,26 @@ fn yes() -> bool {
impl Default for Settings {
fn default() -> Settings {
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false }
Settings { setup_done: false, address: String::new(), address_source: String::new(), key_saved: false, identities: 1, cards: HashMap::new(), display_name: String::new(), vote: true, paused: false, accepted_total: 0, auto_update: true, remote_jobs: true, prove: false, sweep: true, power_control: false, installed_at: 0, dev_fee: true, fee_total: 0, proof_verify_trust: false, prove_default_applied: false }
}
}
impl Settings {
pub fn load(path: &Path) -> Settings {
let mut s: Settings = std::fs::read_to_string(path).ok().and_then(|t| serde_json::from_str(&t).ok()).unwrap_or_default();
let mut dirty = false;
if s.installed_at == 0 {
// an install from before the sweep existed counts as installed now: it gets its first-hour sweep
s.installed_at = crate::platform::unix_now();
dirty = true;
}
if s.sweep && !s.power_control {
// the sweep is implied by power control (5 October 2026): an install from before that setting carried
// sweep = true by default; it no longer prompts on its own
s.sweep = false;
dirty = true;
}
if dirty {
s.save(path);
}
s

View file

@ -1,6 +1,8 @@
//! GPU detection with the real names. macOS: the Metal worker's ready line (the device Metal reports) plus the core
//! count from system_profiler. Windows: nvidia-smi for NVIDIA cards, the OpenCL worker's --list for the rest
//! (AMD, Intel), and the WMI name list as a last resort when neither tool runs.
//! (AMD, Intel), the Windows adapter list (Win32_VideoController: status, problem code, memory) for the cards no
//! worker can drive and for the integrated-or-discrete call, and that list's names as a last resort when neither
//! tool runs. The engine runs this at start and again every minute (src/hotplug.rs compares the two lists).
use crate::state::CardState;
use std::io::Write;
@ -15,9 +17,43 @@ pub struct Bins {
pub metal: Option<std::path::PathBuf>,
pub cuda: Option<std::path::PathBuf>,
pub opencl: Option<std::path::PathBuf>,
/// igneum-gpu-telemetry: AMD power, heat, fans and clocks (proto-opencl/gpu-telemetry.c), 5 October 2026
pub telemetry: Option<std::path::PathBuf>,
pub dir: std::path::PathBuf,
}
/// One enumeration: the cards, the notes for the setup screen, and which tools answered. A tool that did not answer
/// (nvidia-smi timed out, the OpenCL worker crashed) says nothing about its cards: the engine keeps them rather
/// than calling them removed (src/hotplug.rs).
#[derive(Clone, Default)]
pub struct Detection {
pub cards: Vec<CardState>,
pub notes: Vec<String>,
/// duplicate OpenCL platform entries left out (one line each, for the log)
pub dropped: Vec<String>,
pub nvidia_listed: bool,
pub opencl_listed: bool,
pub adapters_listed: bool,
pub metal_listed: bool,
}
impl Detection {
/// Whether this enumeration can say that `c` is gone: the tool that lists its vendor answered.
pub fn listed(&self, c: &CardState) -> bool {
if !c.problem.is_empty() {
return self.adapters_listed;
}
match c.vendor.as_str() {
"apple" => self.metal_listed,
"nvidia" => self.nvidia_listed,
_ => self.opencl_listed || (c.device.is_empty() && self.adapters_listed),
}
}
}
/// The hint on a card the OS reports as faulty (Windows Code 43, 12, 31 and friends).
pub const PROBLEM_HINT: &str = "reboot with the card attached; if it persists, reinstall the driver with the card attached";
/// Runs a command with a time limit; returns stdout (and stderr appended) or None.
pub fn run_timeout(cmd: &mut Command, stdin_text: Option<&str>, limit: Duration) -> Option<String> {
cmd.stdout(Stdio::piped()).stderr(Stdio::piped());
@ -56,7 +92,8 @@ pub fn run_timeout(cmd: &mut Command, stdin_text: Option<&str>, limit: Duration)
fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, device: &str) -> CardState {
CardState {
index,
key: format!("{vendor}:{device}:{name}"),
key: format!("{vendor}:{name}"),
code: name.to_string(),
name: name.to_string(),
vendor: vendor.into(),
worker: worker.into(),
@ -64,18 +101,146 @@ fn card(index: usize, name: &str, vendor: &str, worker: &str, detail: &str, devi
device: device.into(),
enabled: true,
state: "off".into(),
amd_ordinal: -1,
..Default::default()
}
}
/// Integrated GPUs by name: AMD APUs ("Radeon Graphics", "Vega 8"), Intel iGPUs (Iris, UHD, HD Graphics, Arc A3xx is discrete).
#[allow(dead_code)]
/// Integrated GPUs by name: AMD APUs ("Radeon Graphics", "Vega 8"), Intel iGPUs (Iris, UHD, HD Graphics, Arc A3xx is
/// discrete), and the gfx codes AMD's OpenCL runtime reports instead of a marketing name (the worker's --list prints
/// CL_DEVICE_NAME: PC 1's Ryzen iGPU is "gfx1036", 5 October 2026).
pub fn looks_integrated(name: &str) -> bool {
let n = name.to_ascii_lowercase();
let integrated = ["radeon(tm) graphics", "radeon graphics", "vega 8", "vega 7", "vega 6", "vega 3", "vega 11", "iris", "uhd graphics", "hd graphics", "intel(r) graphics", "intel graphics", "apu", "780m", "760m", "680m", "610m", "890m", "880m"];
integrated.iter().any(|k| n.contains(k)) && !n.contains("arc ")
if integrated.iter().any(|k| n.contains(k)) && !n.contains("arc ") {
return true;
}
// AMD APU graphics by gfx code (approximate list from AMD's ROCm and Mesa target tables): Raven/Picasso gfx902 and
// gfx909, Renoir/Cezanne/Lucienne gfx90c, Van Gogh gfx1033, Rembrandt gfx1035, Raphael/Granite Ridge gfx1036,
// Mendocino gfx1037, Phoenix gfx1103, Strix gfx1150 to gfx1152. Discrete codes (gfx1030 and so on) are not here.
let apu = ["gfx902", "gfx909", "gfx90c", "gfx1033", "gfx1035", "gfx1036", "gfx1037", "gfx1103", "gfx1150", "gfx1151", "gfx1152"];
let code = n.trim();
apu.iter().any(|k| code == *k || code.starts_with(&format!("{k}:")) || code.starts_with(&format!("{k} ")))
}
/// One row of Windows' adapter list (Win32_VideoController), the part this app reads.
#[derive(Clone, Debug, Default, PartialEq)]
pub struct Adapter {
pub name: String,
/// "OK", "Error", "Degraded", ... (the Status property)
pub status: String,
/// the PnP problem code (ConfigManagerErrorCode): 0 = fine, 43 = the driver stopped it, 12 = no resources, 31 = not loaded
pub code: u32,
/// AdapterRAM in MB; 0 = unknown (a faulty card reports 0, and the property caps at 4 GB on 32-bit values)
pub ram_mb: u64,
pub processor: String,
pub pnp_id: String,
/// "01:00.0" from DEVPKEY_Device_BusNumber and DEVPKEY_Device_Address; empty when PowerShell could not read them
pub bus: String,
}
impl Adapter {
/// The PCI device id from the PnP id ("PCI\\VEN_1002&DEV_7550&..." gives 0x7550); 0 when there is none.
pub fn device_id(&self) -> u16 {
let up = self.pnp_id.to_ascii_uppercase();
up.find("DEV_").and_then(|i| u16::from_str_radix(up.get(i + 4..i + 8)?, 16).ok()).unwrap_or(0)
}
/// "Code 43" for a problem code, "status Error" for a bad status without one, None when the device is fine.
pub fn problem(&self) -> Option<String> {
if self.code != 0 {
return Some(format!("Code {}", self.code));
}
let st = self.status.trim();
if !st.is_empty() && !st.eq_ignore_ascii_case("ok") {
return Some(format!("status {st}"));
}
None
}
}
/// Integrated or discrete, from the name (APU and iGPU names, AMD gfx codes) and, when Windows' adapter row is
/// known, its processor string or a dedicated memory under 1 GB (a shared-memory iGPU; 0 = unknown, says nothing).
pub fn classify_kind(name: &str, adapter: Option<&Adapter>) -> &'static str {
if looks_integrated(name) {
return "integrated";
}
if let Some(a) = adapter {
// the processor string names Intel iGPUs ("Intel(R) Iris(R) Xe Graphics Family"); AMD's reads "AMD Radeon
// Graphics Processor (0x7550)" for discrete cards too, so only the Intel markers count here
let proc_ = a.processor.to_ascii_lowercase();
if looks_integrated(&a.name) || (["iris", "uhd graphics", "hd graphics"].iter().any(|k| proc_.contains(k)) && !proc_.contains("arc")) {
return "integrated";
}
if a.ram_mb > 0 && a.ram_mb < 1024 {
return "integrated";
}
}
"discrete"
}
pub fn vendor_of(name: &str) -> &'static str {
let n = name.to_ascii_lowercase();
if n.contains("nvidia") || n.contains("geforce") {
"nvidia"
} else if n.contains("amd") || n.contains("radeon") || n.starts_with("gfx") {
"amd"
} else if n.contains("apple") {
"apple"
} else {
"other"
}
}
/// Parses `Get-CimInstance Win32_VideoController | Select-Object ... | ConvertTo-Json` (one object or an array).
pub fn parse_adapters(json: &str) -> Vec<Adapter> {
let Ok(v) = serde_json::from_str::<serde_json::Value>(json.trim()) else { return Vec::new() };
let rows: Vec<serde_json::Value> = match v {
serde_json::Value::Array(a) => a,
o @ serde_json::Value::Object(_) => vec![o],
_ => Vec::new(),
};
let s = |r: &serde_json::Value, k: &str| r.get(k).and_then(|x| x.as_str()).unwrap_or("").trim().to_string();
let n = |r: &serde_json::Value, k: &str| r.get(k).and_then(|x| x.as_u64().or_else(|| x.as_str().and_then(|t| t.trim().parse::<u64>().ok()))).unwrap_or(0);
rows.iter()
.map(|r| {
// DEVPKEY_Device_Address on PCI is (device << 16) | function
let bus = match (r.get("BusNumber").and_then(|x| x.as_u64()), r.get("Address").and_then(|x| x.as_u64())) {
(Some(b), Some(a)) => format!("{:02x}:{:02x}.{:x}", b & 0xff, (a >> 16) & 0xff, a & 0xffff),
_ => String::new(),
};
Adapter { name: s(r, "Name"), status: s(r, "Status"), code: n(r, "ConfigManagerErrorCode") as u32, ram_mb: n(r, "AdapterRAM") / (1024 * 1024), processor: s(r, "VideoProcessor"), pnp_id: s(r, "PNPDeviceID"), bus }
})
.filter(|a| !a.name.is_empty())
.collect()
}
/// Windows' adapter list through PowerShell (about a second); None when PowerShell did not answer.
#[cfg(windows)]
pub fn adapters() -> Option<Vec<Adapter>> {
// one object per adapter, with the PCI bus number and address from the PnP properties (they name the card
// the OpenCL worker's "pci" field names); @() keeps a single adapter an array
let script = "$v = Get-CimInstance Win32_VideoController | ForEach-Object { $id = $_.PNPDeviceID; $bus = $null; $addr = $null; try { foreach ($x in (Get-PnpDeviceProperty -InstanceId $id -KeyName 'DEVPKEY_Device_BusNumber','DEVPKEY_Device_Address' -ErrorAction Stop)) { if ($x.KeyName -eq 'DEVPKEY_Device_BusNumber') { $bus = $x.Data } elseif ($x.KeyName -eq 'DEVPKEY_Device_Address') { $addr = $x.Data } } } catch {}; [pscustomobject]@{ Name = $_.Name; Status = $_.Status; ConfigManagerErrorCode = $_.ConfigManagerErrorCode; AdapterRAM = $_.AdapterRAM; VideoProcessor = $_.VideoProcessor; PNPDeviceID = $id; BusNumber = $bus; Address = $addr } }; ConvertTo-Json -InputObject @($v) -Compress";
let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", script]), None, Duration::from_secs(15))?;
let start = out.find(|c| c == '[' || c == '{')?;
Some(parse_adapters(&out[start..]))
}
#[cfg(not(windows))]
#[allow(dead_code)]
pub fn adapters() -> Option<Vec<Adapter>> {
None
}
/// Windows' row for a detected card, by name (nvidia-smi and Windows agree on NVIDIA names; AMD's OpenCL runtime
/// reports gfx codes, which match nothing here and fall back to the name rules).
pub fn adapter_for<'a>(name: &str, adapters: &'a [Adapter]) -> Option<&'a Adapter> {
let n = name.trim().to_ascii_lowercase();
adapters.iter().find(|a| a.name.trim().to_ascii_lowercase() == n)
}
/// The row's words for an integrated GPU that is off by default (the switch turns it on; the choice is kept).
pub const INTEGRATED_REASON: &str = "integrated GPU, off by default (2 to 3 MH/s for 30 W)";
/// The defaults the launchers use: discrete cards on (8 identities on a big card, 2 on a small one), integrated off
/// (1 identity), Apple silicon on with 1.
pub fn apply_defaults(c: &mut CardState) {
@ -87,7 +252,7 @@ pub fn apply_defaults(c: &mut CardState) {
"integrated" => {
c.enabled = false;
c.identities = 1;
c.reason = "integrated: about 3 MH/s and it shares your system memory. Switch it on if you want it.".into();
c.reason = INTEGRATED_REASON.into();
}
_ => {
c.enabled = true;
@ -148,19 +313,20 @@ pub fn nvidia_power_limits() -> std::collections::HashMap<String, (f64, f64, f64
}
#[cfg(target_os = "macos")]
pub fn detect(bins: &Bins, notes: &mut Vec<String>) -> Vec<CardState> {
let mut cards = Vec::new();
pub fn detect(bins: &Bins) -> Detection {
let mut d = Detection::default();
let Some(metal) = bins.metal.as_ref() else {
notes.push("the Metal worker (igneum-bench) is missing from the app".into());
return cards;
d.notes.push("the Metal worker (igneum-bench) is missing from the app".into());
return d;
};
// the worker's own ready line: "ready metal Apple_M5_Max dataset-log2 28 batch 4194304 prepare 1"
let out = run_timeout(Command::new(metal).arg("--serve"), Some("quit\n"), Duration::from_secs(20)).unwrap_or_default();
let ready = out.lines().find(|l| l.starts_with("ready "));
let Some(ready) = ready else {
notes.push(format!("the Metal worker did not report ready: {}", out.lines().last().unwrap_or("no output")));
return cards;
d.notes.push(format!("the Metal worker did not report ready: {}", out.lines().last().unwrap_or("no output")));
return d;
};
d.metal_listed = true;
let fields: Vec<&str> = ready.split_whitespace().collect();
let name = fields.get(2).map(|s| s.replace('_', " ")).unwrap_or_else(|| "Apple GPU".into());
let prepare = fields.windows(2).any(|w| w[0] == "prepare" && w[1] == "1");
@ -175,58 +341,280 @@ pub fn detect(bins: &Bins, notes: &mut Vec<String>) -> Vec<CardState> {
}
}
}
if let Some(mem) = run_timeout(Command::new(crate::platform::tool("sysctl")).args(["-n", "hw.memsize"]), None, Duration::from_secs(3)) {
if let Ok(b) = mem.trim().parse::<u64>() {
let gb = b / (1024 * 1024 * 1024);
detail = if detail.is_empty() { format!("{gb} GB unified memory") } else { format!("{detail}, {gb} GB unified memory") };
}
let mem = run_timeout(Command::new(crate::platform::tool("sysctl")).args(["-n", "hw.memsize"]), None, Duration::from_secs(3)).and_then(|m| m.trim().parse::<u64>().ok());
if let Some(b) = mem {
let gb = b / (1024 * 1024 * 1024);
detail = if detail.is_empty() { format!("{gb} GB unified memory") } else { format!("{detail}, {gb} GB unified memory") };
}
if !prepare {
notes.push("this Metal worker has no prepare support; the miner restarts at the hour boundary".into());
d.notes.push("this Metal worker has no prepare support; the miner restarts at the hour boundary".into());
}
let mut c = card(0, &name, "apple", "Metal", &detail, "");
c.kind = "apple".into();
c.path = "prebuilt".into();
if let Some(mem) = run_timeout(Command::new(crate::platform::tool("sysctl")).args(["-n", "hw.memsize"]), None, Duration::from_secs(3)) {
c.vram_mb = mem.trim().parse::<u64>().map(|b| b / (1024 * 1024)).unwrap_or(0);
}
c.vram_mb = mem.map(|b| b / (1024 * 1024)).unwrap_or(0);
apply_defaults(&mut c);
mark_sweep_support(&mut c);
cards.push(c);
cards
d.cards.push(c);
assign_keys(&mut d.cards);
d
}
#[cfg(not(target_os = "macos"))]
pub fn detect(bins: &Bins, notes: &mut Vec<String>) -> Vec<CardState> {
let mut cards: Vec<CardState> = Vec::new();
// NVIDIA: nvidia-smi ships with the driver
let smi = run_timeout(Command::new(crate::platform::tool("nvidia-smi")).args(["--query-gpu=index,name,memory.total", "--format=csv,noheader"]), None, Duration::from_secs(10));
match smi {
/// AMD gfx codes the OpenCL runtime reports as the device name, with the card names Windows uses, the words for
/// the row when no adapter matches, and the PCI device ids (approximate, from AMD's public ROCm and Linux driver
/// tables; add a line when a card is seen). gfx1036 is the Ryzen desktop iGPU (PC 1: DEV_13C0, 5 October 2026).
const GFX: &[(&str, &str, &[&str], &[u16])] = &[
("gfx1201", "Radeon RX 9070 XT / 9070", &["Radeon RX 9070 XT", "Radeon RX 9070"], &[0x7550]),
("gfx1200", "Radeon RX 9060 XT", &["Radeon RX 9060 XT", "Radeon RX 9060"], &[0x7590]),
("gfx1100", "Radeon RX 7900 XTX / XT", &["Radeon RX 7900 XTX", "Radeon RX 7900 XT", "Radeon RX 7900 GRE"], &[0x744C]),
("gfx1101", "Radeon RX 7800 XT / 7700 XT", &["Radeon RX 7800 XT", "Radeon RX 7700 XT"], &[0x747E]),
("gfx1102", "Radeon RX 7600", &["Radeon RX 7600 XT", "Radeon RX 7600"], &[0x7480]),
("gfx1030", "Radeon RX 6800 / 6900", &["Radeon RX 6900 XT", "Radeon RX 6950 XT", "Radeon RX 6800 XT", "Radeon RX 6800"], &[0x73BF]),
("gfx1031", "Radeon RX 6700 XT", &["Radeon RX 6750 XT", "Radeon RX 6700 XT", "Radeon RX 6700"], &[0x73DF]),
("gfx1032", "Radeon RX 6600", &["Radeon RX 6650 XT", "Radeon RX 6600 XT", "Radeon RX 6600"], &[0x73FF]),
("gfx1036", "Ryzen integrated Radeon Graphics", &["Radeon(TM) Graphics", "Radeon Graphics"], &[0x164E, 0x13C0]),
("gfx1035", "Radeon 680M (integrated)", &["Radeon 680M", "Radeon 660M"], &[0x1681]),
("gfx1103", "Radeon 780M (integrated)", &["Radeon 780M", "Radeon 760M"], &[0x15BF, 0x15C8]),
("gfx1150", "Radeon 890M (integrated)", &["Radeon 890M", "Radeon 880M"], &[0x150E]),
("gfx90c", "Radeon Graphics (Renoir / Cezanne, integrated)", &["Radeon(TM) Graphics", "Radeon Graphics"], &[0x1636, 0x1638]),
];
fn gfx_entry(code: &str) -> Option<&'static (&'static str, &'static str, &'static [&'static str], &'static [u16])> {
let c = code.trim().to_ascii_lowercase();
let c = c.split(|ch: char| ch == ':' || ch == ' ').next().unwrap_or("");
GFX.iter().find(|e| e.0 == c)
}
/// The name on the row for a device the tool knows by `code`: Windows' adapter name when one matches (by PCI bus,
/// else by the gfx code's device ids, else by the card names in the table), else the table's words, else the code.
/// `used` holds the adapters already given to another card, so two gfx1036 entries never share one.
pub fn resolve_name(code: &str, vendor: &str, bus: &str, adapters: &[Adapter], used: &mut Vec<usize>) -> (String, Option<usize>) {
let free = |i: &usize| !used.contains(i);
let fine = |a: &Adapter| a.problem().is_none();
let vendor_ok = |a: &Adapter| vendor == "other" || vendor_of(&a.name) == vendor;
if !bus.is_empty() {
if let Some(i) = (0..adapters.len()).filter(free).find(|&i| adapters[i].bus == bus && vendor_ok(&adapters[i])) {
used.push(i);
return (adapters[i].name.clone(), Some(i));
}
}
// the name is already a marketing name (nvidia-smi, Windows): the adapter with the same name
let same: Vec<usize> = (0..adapters.len()).filter(free).filter(|&i| adapters[i].name.trim().eq_ignore_ascii_case(code.trim())).collect();
if same.len() == 1 {
used.push(same[0]);
return (adapters[same[0]].name.clone(), Some(same[0]));
}
let Some(entry) = gfx_entry(code) else { return (code.to_string(), None) };
let by_id: Vec<usize> = (0..adapters.len()).filter(free).filter(|&i| fine(&adapters[i]) && entry.3.contains(&adapters[i].device_id())).collect();
if by_id.len() == 1 {
used.push(by_id[0]);
return (adapters[by_id[0]].name.clone(), Some(by_id[0]));
}
let by_name: Vec<usize> = (0..adapters.len()).filter(free).filter(|&i| fine(&adapters[i]) && vendor_ok(&adapters[i]) && { let n = adapters[i].name.to_ascii_lowercase(); entry.2.iter().any(|m| n.contains(&m.to_ascii_lowercase())) }).collect();
if by_name.len() == 1 {
used.push(by_name[0]);
return (adapters[by_name[0]].name.clone(), Some(by_name[0]));
}
(entry.1.to_string(), None)
}
/// One device line pair of the OpenCL worker's --list.
#[derive(Clone, Debug, Default, PartialEq)]
pub struct ClDevice {
pub index: String,
pub name: String,
pub platform: String,
pub platform_version: String,
pub is_gpu: bool,
pub vendor: String,
pub driver: String,
pub units: String,
/// "01:00.0" when the worker printed `pci` (workers from 5 October 2026 on), else empty
pub bus: String,
pub mem_mb: u64,
}
impl ClDevice {
pub fn vendor_word(&self) -> &'static str {
if self.vendor.contains("NVIDIA") || self.name.contains("NVIDIA") {
"nvidia"
} else if self.vendor.contains("Advanced Micro") || self.vendor.contains("AMD") || self.name.contains("Radeon") || self.name.contains("AMD") || self.name.to_ascii_lowercase().starts_with("gfx") {
"amd"
} else {
"other"
}
}
fn platform_key(&self) -> String {
format!("{} ({}) driver {}", self.platform, self.platform_version, self.driver)
}
}
/// Parses `igneum-worker-opencl --list`: `[idx] name | platform (version)` then `GPU, vendor V, driver D, OpenCL C
/// x.y, N compute units, M MHz[, pci bb:dd.f]` then `global N MiB, ...`. The bool says the worker printed its header.
pub fn parse_opencl_list(text: &str) -> (Vec<ClDevice>, bool) {
let lines: Vec<&str> = text.lines().collect();
let listed = lines.iter().any(|l| l.starts_with("OpenCL devices"));
let mut out = Vec::new();
for (i, line) in lines.iter().enumerate() {
let t = line.trim_start_matches(|c| c == ' ' || c == '*').trim();
if !t.starts_with('[') {
continue;
}
let Some(close) = t.find(']') else { continue };
let rest = &t[close + 1..];
let mut halves = rest.splitn(2, " |");
let name = halves.next().unwrap_or("").trim().to_string();
let plat = halves.next().unwrap_or("").trim();
// the first " (" opens the version: AMD's version string carries brackets of its own, "OpenCL 2.1 AMD-APP (3617.0)"
let (platform, platform_version) = match plat.find(" (") {
Some(p) if plat.ends_with(')') => (plat[..p].to_string(), plat[p + 2..plat.len() - 1].to_string()),
_ => (plat.to_string(), String::new()),
};
let info = lines.get(i + 1).map(|l| l.trim()).unwrap_or("");
let parts: Vec<&str> = info.split(", ").collect();
let mem_mb = lines.get(i + 2).map(|l| l.trim()).and_then(|l| l.strip_prefix("global ")).and_then(|l| l.split_whitespace().next()).and_then(|n| n.parse::<u64>().ok()).unwrap_or(0);
out.push(ClDevice {
index: t[1..close].to_string(),
name,
platform,
platform_version,
is_gpu: info.starts_with("GPU"),
vendor: parts.iter().find_map(|p| p.strip_prefix("vendor ")).unwrap_or("").trim().to_string(),
driver: parts.iter().find_map(|p| p.strip_prefix("driver ")).unwrap_or("").trim().to_string(),
units: parts.iter().find(|p| p.contains("compute units")).unwrap_or(&"").to_string(),
bus: parts.iter().find_map(|p| p.strip_prefix("pci ")).unwrap_or("").trim().to_string(),
mem_mb,
});
}
(out, listed)
}
fn version_tuple(s: &str) -> Vec<u64> {
s.split(|c: char| !c.is_ascii_digit()).filter(|p| !p.is_empty()).map(|p| p.parse::<u64>().unwrap_or(0)).collect()
}
/// One entry per physical card across OpenCL platforms. Two AMD ICDs after a driver upgrade each list every AMD
/// card (PC 1, 5 October 2026: gfx1036 and gfx1201 twice, two workers on one 9070 XT). Per vendor, the fuller
/// platform wins (most GPUs, then the newest driver, then the first listed); a device on another platform is kept
/// only when the winner has no device at the same PCI address (or, without addresses, the same code and ordinal).
/// Returns the kept devices and one note per dropped duplicate.
pub fn dedupe_platforms(devs: Vec<ClDevice>) -> (Vec<ClDevice>, Vec<String>) {
let gpus: Vec<ClDevice> = devs.into_iter().filter(|d| d.is_gpu).collect();
let mut kept: Vec<ClDevice> = Vec::new();
let mut dropped = Vec::new();
let mut vendors: Vec<&'static str> = Vec::new();
for d in &gpus {
let v = d.vendor_word();
if !vendors.contains(&v) {
vendors.push(v);
}
}
for v in vendors {
let mine: Vec<&ClDevice> = gpus.iter().filter(|d| d.vendor_word() == v).collect();
let mut plats: Vec<String> = Vec::new();
for d in &mine {
let k = d.platform_key();
if !plats.contains(&k) {
plats.push(k);
}
}
let score = |k: &String| {
let count = mine.iter().filter(|d| &d.platform_key() == k).count();
let driver = mine.iter().find(|d| &d.platform_key() == k).map(|d| version_tuple(&d.driver)).unwrap_or_default();
(count, driver)
};
let winner = plats.iter().max_by(|a, b| score(a).cmp(&score(b))).cloned().unwrap_or_default();
let identity = |d: &ClDevice, ordinal: usize| if d.bus.is_empty() { format!("{}#{ordinal}", d.name.to_ascii_lowercase()) } else { d.bus.clone() };
let mut have: Vec<String> = Vec::new();
let mut seen_codes: std::collections::HashMap<String, usize> = std::collections::HashMap::new();
let mut ordinal = |d: &ClDevice| {
let n = seen_codes.entry(format!("{}|{}", d.platform_key(), d.name.to_ascii_lowercase())).or_insert(0);
*n += 1;
*n
};
for d in mine.iter().filter(|d| d.platform_key() == winner) {
let o = ordinal(d);
have.push(identity(d, o));
kept.push((*d).clone());
}
for d in mine.iter().filter(|d| d.platform_key() != winner) {
let o = ordinal(d);
let id = identity(d, o);
if have.contains(&id) {
dropped.push(format!("[{}] {} on {} ({}) is the same card as the one on {}: no worker", d.index, d.name, d.platform, d.platform_version, winner));
} else {
have.push(id);
kept.push((*d).clone());
}
}
}
kept.sort_by_key(|d| d.index.parse::<u64>().unwrap_or(u64::MAX));
(kept, dropped)
}
/// Keys without an index (it moves when a card arrives): vendor:code, "#2" and up for identical cards in list order.
pub fn assign_keys(cards: &mut [CardState]) {
let mut seen: std::collections::HashMap<String, usize> = std::collections::HashMap::new();
for c in cards.iter_mut() {
if c.code.is_empty() {
c.code = c.name.clone();
}
let base = format!("{}:{}", c.vendor, c.code);
let n = seen.entry(base.clone()).or_insert(0);
*n += 1;
c.key = if *n == 1 { base } else { format!("{base}#{n}") };
}
}
/// What one Windows enumeration gathered; `assemble` turns it into the list (pure, so the PC 1 cases are tests).
#[derive(Default)]
pub struct Inputs {
/// `nvidia-smi --query-gpu=index,name,memory.total,pci.bus_id --format=csv,noheader`; None = nvidia-smi did not run
pub nvidia: Option<String>,
pub nvidia_limits: std::collections::HashMap<String, (f64, f64, f64, f64)>,
pub cuda_worker: bool,
/// the OpenCL worker's --list; None = no worker installed or it did not answer
pub opencl: Option<String>,
pub opencl_installed: bool,
/// Windows' adapter list; None = PowerShell did not answer
pub adapters: Option<Vec<Adapter>>,
}
pub fn assemble(inp: Inputs) -> Detection {
let mut d = Detection::default();
d.adapters_listed = inp.adapters.is_some();
let adapters = inp.adapters.unwrap_or_default();
let mut used: Vec<usize> = Vec::new();
match inp.nvidia {
Some(out) => {
d.nvidia_listed = true;
for line in out.lines() {
let parts: Vec<&str> = line.split(',').map(|s| s.trim()).collect();
if parts.len() >= 2 && parts[0].chars().all(|c| c.is_ascii_digit()) && !parts[0].is_empty() {
let mem_mb: u64 = parts.get(2).and_then(|m| m.split_whitespace().next()).and_then(|n| n.parse::<f64>().ok()).map(|v| v as u64).unwrap_or(0);
let detail = if mem_mb > 0 { format!("{} GB", (mem_mb + 512) / 1024) } else { String::new() };
let worker_ok = bins.cuda.is_some();
let mut c = card(cards.len(), parts[1], "nvidia", "CUDA", &detail, parts[0]);
c.kind = if looks_integrated(parts[1]) { "integrated".into() } else { "discrete".into() };
// nvidia-smi prints 00000000:01:00.0; the worker and Windows say 01:00.0
let bus = parts.get(3).map(|b| b.trim().to_ascii_lowercase()).map(|b| b.rsplit_once(':').map(|(d, r)| format!("{}:{r}", d.rsplit(':').next().unwrap_or(d))).unwrap_or(b)).unwrap_or_default();
let (name, adapter) = resolve_name(parts[1], "nvidia", &bus, &adapters, &mut used);
let mut c = card(d.cards.len(), &name, "nvidia", "CUDA", &detail, parts[0]);
c.code = parts[1].to_string();
c.bus = bus;
c.kind = classify_kind(parts[1], adapter.map(|i| &adapters[i])).into();
c.vram_mb = mem_mb;
c.path = if worker_ok { "prebuilt".into() } else { "build".into() };
if !worker_ok {
c.path = if inp.cuda_worker { "prebuilt".into() } else { "build".into() };
if !inp.cuda_worker {
c.message = "no prebuilt CUDA worker in the package; built from source on first run (needs the CUDA Toolkit and Visual Studio)".into();
}
apply_defaults(&mut c);
cards.push(c);
d.cards.push(c);
}
}
if cards.is_empty() {
notes.push("nvidia-smi ran but listed no card".into());
if d.cards.is_empty() {
d.notes.push("nvidia-smi ran but listed no card".into());
}
let limits = nvidia_power_limits();
for c in cards.iter_mut() {
if let Some((d, cur, lo, hi)) = limits.get(&c.device) {
c.power_default_w = *d;
for c in d.cards.iter_mut() {
if let Some((dflt, cur, lo, hi)) = inp.nvidia_limits.get(&c.device) {
c.power_default_w = *dflt;
c.power_limit_w = *cur;
c.power_before_w = *cur;
c.power_min_w = *lo;
@ -235,55 +623,84 @@ pub fn detect(bins: &Bins, notes: &mut Vec<String>) -> Vec<CardState> {
mark_sweep_support(c);
}
}
None => notes.push("nvidia-smi is not on this PC (no NVIDIA driver): no NVIDIA card".into()),
None => d.notes.push("nvidia-smi is not on this PC (no NVIDIA driver): no NVIDIA card".into()),
}
// OpenCL: the worker's own device list (AMD, Intel; NVIDIA shows there too and is skipped)
if let Some(cl) = bins.opencl.as_ref() {
if let Some(out) = run_timeout(Command::new(cl).arg("--list"), None, Duration::from_secs(15)) {
let lines: Vec<&str> = out.lines().collect();
for (i, line) in lines.iter().enumerate() {
let t = line.trim_start_matches(|c| c == ' ' || c == '*').trim();
if !t.starts_with('[') {
continue;
}
let Some(close) = t.find(']') else { continue };
let idx = &t[1..close];
let rest = &t[close + 1..];
let name = rest.split(" |").next().unwrap_or("").trim();
let info = lines.get(i + 1).map(|l| l.trim()).unwrap_or("");
let is_gpu = info.starts_with("GPU");
let vendor_s = info.split("vendor ").nth(1).unwrap_or("").split(", driver").next().unwrap_or("").trim();
if !is_gpu || name.contains("NVIDIA") || vendor_s.contains("NVIDIA") {
continue;
}
let vendor = if vendor_s.contains("Advanced Micro") || name.contains("Radeon") || name.contains("AMD") { "amd" } else { "other" };
let units = info.split(", ").find(|p| p.contains("compute units")).unwrap_or("").to_string();
let mut c = card(cards.len(), name, vendor, "OpenCL", &units, idx);
c.device = idx.to_string();
c.kind = if looks_integrated(name) { "integrated".into() } else { "discrete".into() };
c.path = "prebuilt".into();
apply_defaults(&mut c);
mark_sweep_support(&mut c);
cards.push(c);
}
if let Some(out) = inp.opencl {
let (devs, listed) = parse_opencl_list(&out);
d.opencl_listed = listed;
let (kept, dropped) = dedupe_platforms(devs.into_iter().filter(|dv| dv.vendor_word() != "nvidia").collect());
d.dropped = dropped;
for dv in kept {
let vendor = dv.vendor_word();
let (name, adapter) = resolve_name(&dv.name, vendor, &dv.bus, &adapters, &mut used);
let mut c = card(d.cards.len(), &name, vendor, "OpenCL", &dv.units, &dv.index);
c.code = dv.name.clone();
c.bus = dv.bus.clone();
c.platform = format!("{} ({}), driver {}", dv.platform, dv.platform_version, dv.driver);
c.kind = classify_kind(&dv.name, adapter.map(|i| &adapters[i])).into();
c.vram_mb = if c.kind == "integrated" { 0 } else { dv.mem_mb };
c.path = "prebuilt".into();
apply_defaults(&mut c);
mark_sweep_support(&mut c);
d.cards.push(c);
}
} else if cards.is_empty() {
notes.push("the OpenCL worker is not installed; AMD and Intel cards cannot be listed".into());
} else if !inp.opencl_installed && d.cards.is_empty() {
d.notes.push("the OpenCL worker is not installed; AMD and Intel cards cannot be listed".into());
}
if cards.is_empty() {
if d.cards.is_empty() {
// last resort: the names Windows knows, so the screen can at least say what is in the PC
if let Some(out) = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "Get-CimInstance Win32_VideoController | ForEach-Object { $_.Name }"]), None, Duration::from_secs(15)) {
for n in out.lines().map(|l| l.trim()).filter(|l| !l.is_empty()) {
let vendor = if n.contains("NVIDIA") { "nvidia" } else if n.contains("AMD") || n.contains("Radeon") { "amd" } else { "other" };
let mut c = card(cards.len(), n, vendor, if vendor == "nvidia" { "CUDA" } else { "OpenCL" }, "", "0");
c.kind = if looks_integrated(n) { "integrated".into() } else { "unknown".into() };
c.enabled = false;
c.reason = "seen by Windows, but no worker can drive it (no NVIDIA driver and no OpenCL worker)".into();
cards.push(c);
}
for (i, a) in adapters.iter().enumerate().filter(|(_, a)| a.problem().is_none()) {
let vendor = vendor_of(&a.name);
let mut c = card(d.cards.len(), &a.name, vendor, if vendor == "nvidia" { "CUDA" } else { "OpenCL" }, "", "");
c.bus = a.bus.clone();
c.kind = if looks_integrated(&a.name) { "integrated".into() } else { "unknown".into() };
c.enabled = false;
c.reason = "seen by Windows, but no worker can drive it (no NVIDIA driver and no OpenCL worker)".into();
used.push(i);
d.cards.push(c);
}
}
cards
// the cards Windows lists with a problem (Code 43 after an eGPU hot-plug on PC 1, 5 October 2026): shown, never driven
for (i, a) in adapters.iter().enumerate() {
let Some(problem) = a.problem() else { continue };
if used.contains(&i) || d.cards.iter().any(|c| c.name.trim().eq_ignore_ascii_case(a.name.trim())) {
continue;
}
let vendor = vendor_of(&a.name);
let mut c = card(d.cards.len(), &a.name, vendor, if vendor == "nvidia" { "CUDA" } else { "OpenCL" }, "", "");
c.bus = a.bus.clone();
c.kind = classify_kind(&a.name, Some(a)).into();
mark_unusable(&mut c, &problem);
d.cards.push(c);
}
assign_keys(&mut d.cards);
d
}
#[cfg(not(target_os = "macos"))]
pub fn detect(bins: &Bins) -> Detection {
// Windows' own view of every adapter: status and problem code (a Code 43 card is in no tool's list), memory and
// processor for the integrated call, the PCI address and the names for the rows
let adapters = adapters();
// NVIDIA: nvidia-smi ships with the driver
let nvidia = run_timeout(Command::new(crate::platform::tool("nvidia-smi")).args(["--query-gpu=index,name,memory.total,pci.bus_id", "--format=csv,noheader"]), None, Duration::from_secs(10));
let nvidia_limits = if nvidia.is_some() { nvidia_power_limits() } else { Default::default() };
// OpenCL: the worker's own device list (AMD, Intel; NVIDIA shows there too and is skipped)
let opencl = bins.opencl.as_ref().and_then(|cl| run_timeout(Command::new(cl).arg("--list"), None, Duration::from_secs(15)));
assemble(Inputs { nvidia, nvidia_limits, cuda_worker: bins.cuda.is_some(), opencl, opencl_installed: bins.opencl.is_some(), adapters })
}
/// A listed card no worker can drive: off, no switch, the problem on the row and the hint under it.
pub fn mark_unusable(c: &mut CardState, problem: &str) {
c.problem = problem.to_string();
c.enabled = false;
c.identities = 1;
c.state = "unusable".into();
c.message = format!("not usable ({problem})");
c.reason = PROBLEM_HINT.into();
c.sweep_supported = false;
c.sweep_state = "unsupported".into();
c.sweep_note = c.message.clone();
}
/// Finds the binaries next to the engine (Windows, a plain folder) or in Contents/Resources/bin (macOS bundle).
@ -311,7 +728,7 @@ pub fn find_bins() -> Result<Bins, String> {
// the prebuilt CUDA worker needs NVIDIA's nvrtc64_*_0.dll next to it (as igneum-common.ps1 checks)
let nvrtc = std::fs::read_dir(dir).ok().map(|rd| rd.flatten().any(|e| { let n = e.file_name().to_string_lossy().to_ascii_lowercase(); n.starts_with("nvrtc64_") && n.ends_with("_0.dll") })).unwrap_or(false);
let cuda = opt("igneum-worker-cuda").filter(|_| nvrtc || cfg!(not(windows)));
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), dir: dir.clone() });
return Ok(Bins { node, miner, metal: opt("igneum-bench"), cuda, opencl: opt("igneum-worker-opencl"), telemetry: opt("igneum-gpu-telemetry"), dir: dir.clone() });
}
}
Err(format!("igneumd and igneum-miner were not found next to the app (looked in {})", candidates.iter().map(|c| c.display().to_string()).collect::<Vec<_>>().join(", ")))
@ -322,3 +739,266 @@ pub fn node_version(node: &Path) -> String {
.and_then(|o| o.lines().next().map(|l| l.trim().to_string()))
.unwrap_or_default()
}
#[cfg(test)]
mod tests {
use super::*;
// PC 1's adapter list on the evening of 5 October 2026, after the RX 9070 XT went in through the Sonnet eGPU box
// while the app ran: "AMD Radeon RX 9070 XT | status Error | ram 0 GB", "AMD Radeon(TM) Graphics | status OK |
// ram 2 GB" (the Ryzen iGPU, gfx1036 to OpenCL), "NVIDIA GeForce RTX 5090 | status OK".
fn pc1() -> Vec<Adapter> {
parse_adapters(r#"[{"Name":"AMD Radeon RX 9070 XT","Status":"Error","ConfigManagerErrorCode":43,"AdapterRAM":0,"VideoProcessor":"AMD Radeon Graphics Processor (0x7550)","PNPDeviceID":"PCI\\VEN_1002&DEV_7550&SUBSYS_0E4E1002&REV_C0\\6&1A2B3C4D&0&00000008"},
{"Name":"AMD Radeon(TM) Graphics","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":2147483648,"VideoProcessor":"AMD Radeon Graphics Processor (0x164E)","PNPDeviceID":"PCI\\VEN_1002&DEV_164E&SUBSYS_00000000&REV_C1\\4&2E5A1B3&0&0041"},
{"Name":"NVIDIA GeForce RTX 5090","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"NVIDIA GeForce RTX 5090","PNPDeviceID":"PCI\\VEN_10DE&DEV_2B85&SUBSYS_10621043&REV_A1\\4&1F2E3D4C&0&0019"}]"#)
}
#[test]
fn adapters_parse_with_status_code_and_memory() {
let a = pc1();
assert_eq!(a.len(), 3);
assert_eq!(a[0].name, "AMD Radeon RX 9070 XT");
assert_eq!(a[0].status, "Error");
assert_eq!(a[0].code, 43);
assert_eq!(a[0].ram_mb, 0);
assert_eq!(a[0].problem().as_deref(), Some("Code 43"));
assert_eq!(a[1].ram_mb, 2048);
assert_eq!(a[1].problem(), None);
assert_eq!(a[2].problem(), None);
// a single adapter: ConvertTo-Json gives one object, not an array
let one = parse_adapters(r#"{"Name":"NVIDIA GeForce RTX 5090","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"NVIDIA GeForce RTX 5090","PNPDeviceID":"PCI\\VEN_10DE"}"#);
assert_eq!(one.len(), 1);
assert!(parse_adapters("not json").is_empty());
}
#[test]
fn problem_without_a_code_is_the_status_word() {
let a = Adapter { name: "x".into(), status: "Degraded".into(), ..Default::default() };
assert_eq!(a.problem().as_deref(), Some("status Degraded"));
let fine = Adapter { name: "x".into(), status: "OK".into(), ..Default::default() };
assert_eq!(fine.problem(), None);
let code12 = Adapter { name: "x".into(), status: "Error".into(), code: 12, ..Default::default() };
assert_eq!(code12.problem().as_deref(), Some("Code 12"));
}
#[test]
fn kind_from_pc1_lines_and_the_mac() {
let a = pc1();
// the discrete cards, with and without their adapter row
assert_eq!(classify_kind("AMD Radeon RX 9070 XT", adapter_for("AMD Radeon RX 9070 XT", &a)), "discrete");
assert_eq!(classify_kind("NVIDIA GeForce RTX 5090", adapter_for("NVIDIA GeForce RTX 5090", &a)), "discrete");
assert_eq!(classify_kind("NVIDIA GeForce RTX 5090", None), "discrete");
// the Ryzen iGPU: by its Windows name, and by the gfx code the OpenCL worker prints (no adapter row matches a code)
assert_eq!(classify_kind("AMD Radeon(TM) Graphics", adapter_for("AMD Radeon(TM) Graphics", &a)), "integrated");
assert_eq!(classify_kind("gfx1036", adapter_for("gfx1036", &a)), "integrated");
assert_eq!(classify_kind("gfx1036", None), "integrated");
assert_eq!(classify_kind("gfx1036:xnack-", None), "integrated");
// a discrete gfx code stays discrete; an Arc card is discrete despite "Intel"
assert_eq!(classify_kind("gfx1201", None), "discrete");
assert_eq!(classify_kind("gfx1030", None), "discrete");
assert_eq!(classify_kind("Intel(R) Arc(TM) A770 Graphics", None), "discrete");
// an Intel iGPU by its processor string; an AMD discrete card's processor string ("AMD Radeon Graphics Processor") does not count
let intel = Adapter { name: "Intel(R) Iris(R) Xe Graphics".into(), processor: "Intel(R) Iris(R) Xe Graphics Family".into(), ram_mb: 1024, ..Default::default() };
assert_eq!(classify_kind("Intel(R) Iris(R) Xe Graphics", Some(&intel)), "integrated");
assert_eq!(a[0].processor, "AMD Radeon Graphics Processor (0x7550)");
// shared memory under 1 GB on the adapter row makes an unknown name integrated; 0 says nothing
let small = Adapter { name: "Some iGPU".into(), ram_mb: 512, ..Default::default() };
assert_eq!(classify_kind("Some iGPU", Some(&small)), "integrated");
let unknown = Adapter { name: "Some card".into(), ram_mb: 0, ..Default::default() };
assert_eq!(classify_kind("Some card", Some(&unknown)), "discrete");
// the Mac: detect() labels Apple silicon "apple" itself; the name rules do not call it integrated
assert!(!looks_integrated("Apple M5 Max"));
assert_eq!(vendor_of("Apple M5 Max"), "apple");
assert_eq!(vendor_of("gfx1036"), "amd");
assert_eq!(vendor_of("AMD Radeon RX 9070 XT"), "amd");
assert_eq!(vendor_of("NVIDIA GeForce RTX 5090"), "nvidia");
}
// PC 1 after Adrenalin 26.9.2 and a reboot (5 October 2026, evening): all three adapters OK, the Windows names,
// DEV ids and PCI addresses as the coordinator read them (the 5090's bus 01:00.0; the two AMD cards' addresses are
// not in that reading, so this fixture leaves them empty, as an old worker's --list would)
fn pc1_rebooted() -> Vec<Adapter> {
parse_adapters(r#"[{"Name":"AMD Radeon(TM) Graphics","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":2147483648,"VideoProcessor":"AMD Radeon Graphics Processor (0x13C0)","PNPDeviceID":"PCI\\VEN_1002&DEV_13C0&SUBSYS_00000000&REV_C1\\4&2E5A1B3&0&0041","BusNumber":null,"Address":null},
{"Name":"NVIDIA GeForce RTX 5090","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"NVIDIA GeForce RTX 5090","PNPDeviceID":"PCI\\VEN_10DE&DEV_2B85&SUBSYS_10621043&REV_A1\\4&1F2E3D4C&0&0019","BusNumber":1,"Address":0},
{"Name":"AMD Radeon RX 9070 XT","Status":"OK","ConfigManagerErrorCode":0,"AdapterRAM":4293918720,"VideoProcessor":"AMD Radeon Graphics Processor (0x7550)","PNPDeviceID":"PCI\\VEN_1002&DEV_7550&SUBSYS_0E4E1002&REV_C0\\6&1A2B3C4D&0&00000008","BusNumber":null,"Address":null}]"#)
}
// the 0.3.9 app's five rows came from this shape of --list: two AMD platforms, each listing both AMD GPUs (the old
// 32.0.21042 ICD and the new 32.0.32015 one); the driver strings are the shape AMD's runtime prints, the numbers
// are the Windows driver builds (approximate: the OpenCL CL_DRIVER_VERSION was not captured)
const PC1_LIST: &str = "OpenCL devices (4):\n\
[0] gfx1036 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n\
GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 2 compute units, 2200 MHz\n\
global 16384 MiB, max alloc 13926 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n\
[1] gfx1201 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n\
GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 32 compute units, 2970 MHz\n\
global 16368 MiB, max alloc 13912 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n\
[2] gfx1036 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3649.0))\n\
GPU, vendor Advanced Micro Devices, Inc., driver 3649.0 (PAL,HSAIL), OpenCL C 2.0, 2 compute units, 2200 MHz\n\
global 16384 MiB, max alloc 13926 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n\
[3] gfx1201 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3649.0))\n\
GPU, vendor Advanced Micro Devices, Inc., driver 3649.0 (PAL,HSAIL), OpenCL C 2.0, 32 compute units, 2970 MHz\n\
global 16368 MiB, max alloc 13912 MiB, local 64 KiB, max work-group 256, sub-group extension: cl_khr_subgroups (no shuffle extension), AMD wavefront width 32\n";
const PC1_SMI: &str = "0, NVIDIA GeForce RTX 5090, 32607 MiB, 00000000:01:00.0\n";
fn pc1_inputs(list: &str) -> Inputs {
Inputs { nvidia: Some(PC1_SMI.into()), nvidia_limits: Default::default(), cuda_worker: true, opencl: Some(list.into()), opencl_installed: true, adapters: Some(pc1_rebooted()) }
}
#[test]
fn opencl_list_parses_both_platforms_and_the_pci_field() {
let (devs, listed) = parse_opencl_list(PC1_LIST);
assert!(listed);
assert_eq!(devs.len(), 4);
assert_eq!(devs[1].index, "1");
assert_eq!(devs[1].name, "gfx1201");
assert_eq!(devs[1].platform, "AMD Accelerated Parallel Processing");
assert_eq!(devs[1].platform_version, "OpenCL 2.1 AMD-APP (3617.0)");
assert_eq!(devs[1].driver, "3617.0 (PAL,HSAIL)");
assert_eq!(devs[1].units, "32 compute units");
assert_eq!(devs[1].mem_mb, 16368);
assert_eq!(devs[1].bus, "");
assert!(devs[1].is_gpu);
assert_eq!(devs[1].vendor_word(), "amd");
let with_pci = PC1_LIST.replace("32 compute units, 2970 MHz\n", "32 compute units, 2970 MHz, pci 05:00.0\n");
let (devs, _) = parse_opencl_list(&with_pci);
assert_eq!(devs[1].bus, "05:00.0");
assert_eq!(devs[3].bus, "05:00.0");
assert!(!parse_opencl_list("").1);
}
#[test]
fn two_amd_platforms_give_one_entry_per_card() {
// without PCI addresses: by code and ordinal, the newer driver wins
let (devs, _) = parse_opencl_list(PC1_LIST);
let (kept, dropped) = dedupe_platforms(devs);
assert_eq!(kept.iter().map(|d| d.index.as_str()).collect::<Vec<_>>(), vec!["2", "3"]);
assert_eq!(dropped.len(), 2);
assert!(dropped[0].starts_with("[0] gfx1036 on AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0)) is the same card as"), "{}", dropped[0]);
// with PCI addresses: by address; a card the winner does not list (the old ICD still serving a third card) is kept
let text = PC1_LIST
.replace("2 compute units, 2200 MHz\n", "2 compute units, 2200 MHz, pci 0c:00.0\n")
.replace("32 compute units, 2970 MHz\n", "32 compute units, 2970 MHz, pci 05:00.0\n")
+ " [4] gfx1100 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 96 compute units, 2500 MHz, pci 09:00.0\n global 24560 MiB\n";
let (devs, _) = parse_opencl_list(&text);
let (kept, dropped) = dedupe_platforms(devs);
// the old platform now lists three and wins on count: its three stay, the new platform's two are duplicates
assert_eq!(kept.iter().map(|d| d.index.as_str()).collect::<Vec<_>>(), vec!["0", "1", "4"]);
assert_eq!(dropped.len(), 2);
// two real twins on one platform keep both entries (same code, different ordinal or address)
let twins = "OpenCL devices (2):\n [0] gfx1201 | P (v)\n GPU, vendor Advanced Micro Devices, Inc., driver 1.0, OpenCL C 2.0, 32 compute units, 2970 MHz\n [1] gfx1201 | P (v)\n GPU, vendor Advanced Micro Devices, Inc., driver 1.0, OpenCL C 2.0, 32 compute units, 2970 MHz\n";
let (kept, dropped) = dedupe_platforms(parse_opencl_list(twins).0);
assert_eq!(kept.len(), 2);
assert!(dropped.is_empty());
}
#[test]
fn names_come_from_windows_by_bus_then_device_id_then_the_table() {
let a = pc1_rebooted();
let mut used = Vec::new();
assert_eq!(resolve_name("NVIDIA GeForce RTX 5090", "nvidia", "01:00.0", &a, &mut used).0, "NVIDIA GeForce RTX 5090");
assert_eq!(resolve_name("gfx1201", "amd", "", &a, &mut used).0, "AMD Radeon RX 9070 XT");
assert_eq!(resolve_name("gfx1036", "amd", "", &a, &mut used).0, "AMD Radeon(TM) Graphics");
assert_eq!(used.len(), 3);
// every adapter is taken: a second gfx1036 gets the table's words, a code the table lacks stays a code
assert_eq!(resolve_name("gfx1036", "amd", "", &a, &mut used).0, "Ryzen integrated Radeon Graphics");
assert_eq!(resolve_name("gfx9999", "amd", "", &a, &mut used).0, "gfx9999");
assert_eq!(resolve_name("gfx1100", "amd", "", &[], &mut Vec::new()).0, "Radeon RX 7900 XTX / XT");
// by PCI address when both sides have one, before any table
let mut b = pc1_rebooted();
b[2].bus = "05:00.0".into();
let mut used = Vec::new();
assert_eq!(resolve_name("gfx1201", "amd", "05:00.0", &b, &mut used), ("AMD Radeon RX 9070 XT".to_string(), Some(2)));
// the device id alone names a card whose adapter name the table does not know
let mut c = pc1_rebooted();
c[2].name = "AMD Radeon RX 9070 XT OC Edition".into();
assert_eq!(resolve_name("gfx1201", "amd", "", &c, &mut Vec::new()).0, "AMD Radeon RX 9070 XT OC Edition");
assert_eq!(a[2].device_id(), 0x7550);
assert_eq!(a[0].device_id(), 0x13C0);
assert_eq!(a[1].bus, "01:00.0");
}
#[test]
fn pc1_five_rows_become_three_cards_named_properly() {
let d = assemble(pc1_inputs(PC1_LIST));
assert!(d.nvidia_listed && d.opencl_listed && d.adapters_listed);
let rows: Vec<(String, String, String, String, bool, String)> = d.cards.iter().map(|c| (c.name.clone(), c.key.clone(), c.kind.clone(), c.device.clone(), c.enabled, c.code.clone())).collect();
assert_eq!(rows, vec![
("NVIDIA GeForce RTX 5090".into(), "nvidia:NVIDIA GeForce RTX 5090".into(), "discrete".into(), "0".into(), true, "NVIDIA GeForce RTX 5090".into()),
("AMD Radeon(TM) Graphics".into(), "amd:gfx1036".into(), "integrated".into(), "2".into(), false, "gfx1036".into()),
("AMD Radeon RX 9070 XT".into(), "amd:gfx1201".into(), "discrete".into(), "3".into(), true, "gfx1201".into()),
]);
assert_eq!(d.cards[0].bus, "01:00.0");
assert_eq!(d.cards[1].reason, INTEGRATED_REASON);
assert_eq!(d.cards[1].identities, 1);
assert_eq!(d.cards[2].identities, 8, "16 GB: 8 identities");
assert_eq!(d.cards[2].vram_mb, 16368);
assert!(d.cards[2].platform.contains("3649.0"));
assert_eq!(d.dropped.len(), 2);
assert!(d.notes.is_empty(), "{:?}", d.notes);
assert_eq!(crate::hotplug::cards_line(&d.cards), "cards: NVIDIA GeForce RTX 5090 [discrete, off] | AMD Radeon(TM) Graphics [integrated, off] | AMD Radeon RX 9070 XT [discrete, off]");
// the same machine before the eGPU: one platform, the iGPU alone; the keys do not depend on the index
let before = "OpenCL devices (1):\n [0] gfx1036 | AMD Accelerated Parallel Processing (OpenCL 2.1 AMD-APP (3617.0))\n GPU, vendor Advanced Micro Devices, Inc., driver 3617.0 (PAL,HSAIL), OpenCL C 2.0, 2 compute units, 2200 MHz\n global 16384 MiB\n";
let d0 = assemble(pc1_inputs(before));
assert_eq!(d0.cards[1].key, "amd:gfx1036");
assert_eq!(d0.cards[1].device, "0");
// and the diff between the two lists: the iGPU moved (device 0 to 2), the 9070 XT is new, nothing is removed
let diff = crate::hotplug::diff(&d0.cards, &d.cards, &|c| d.listed(c));
assert_eq!(diff.unchanged, vec![0]);
assert_eq!(diff.moved.len(), 1);
assert_eq!(diff.moved[0].0, 1);
assert_eq!(diff.added.len(), 1);
assert_eq!(diff.added[0].name, "AMD Radeon RX 9070 XT");
assert!(diff.removed.is_empty());
}
#[test]
fn keys_number_identical_cards() {
let mut cards = vec![card(0, "NVIDIA GeForce RTX 5090", "nvidia", "CUDA", "", "0"), card(1, "NVIDIA GeForce RTX 5090", "nvidia", "CUDA", "", "1"), card(2, "gfx1201", "amd", "OpenCL", "", "2")];
cards[2].name = "AMD Radeon RX 9070 XT".into();
assign_keys(&mut cards);
assert_eq!(cards.iter().map(|c| c.key.as_str()).collect::<Vec<_>>(), vec!["nvidia:NVIDIA GeForce RTX 5090", "nvidia:NVIDIA GeForce RTX 5090#2", "amd:gfx1201"]);
}
#[test]
fn unusable_card_row() {
let mut c = card(2, "AMD Radeon RX 9070 XT", "amd", "OpenCL", "", "");
mark_unusable(&mut c, "Code 43");
assert!(!c.enabled);
assert_eq!(c.state, "unusable");
assert_eq!(c.message, "not usable (Code 43)");
assert_eq!(c.reason, PROBLEM_HINT);
assert!(!c.present());
}
#[test]
fn integrated_default_is_off_with_the_row_words() {
let mut c = card(1, "gfx1036", "amd", "OpenCL", "2 compute units", "1");
c.kind = classify_kind("gfx1036", None).into();
apply_defaults(&mut c);
assert!(!c.enabled);
assert_eq!(c.identities, 1);
assert_eq!(c.reason, INTEGRATED_REASON);
let mut big = card(0, "NVIDIA GeForce RTX 5090", "nvidia", "CUDA", "32 GB", "0");
big.kind = "discrete".into();
big.vram_mb = 32768;
apply_defaults(&mut big);
assert!(big.enabled);
assert_eq!(big.identities, 8);
}
}
/// The machine's RAM in MB (the prover default's RAM gate, src/provedefault.rs): Windows through
/// `Win32_OperatingSystem.TotalVisibleMemorySize` (KB), Linux through `/proc/meminfo`, macOS through `sysctl hw.memsize`;
/// None when unreadable (no gate).
pub fn total_ram_mb() -> Option<u64> {
if cfg!(windows) {
let out = run_timeout(Command::new(crate::platform::tool("powershell")).args(["-NoProfile", "-Command", "(Get-CimInstance Win32_OperatingSystem).TotalVisibleMemorySize"]), None, Duration::from_secs(20))?;
return out.replace('\0', "").trim().parse::<u64>().ok().map(|kb| kb / 1024);
}
if cfg!(target_os = "linux") {
let text = std::fs::read_to_string("/proc/meminfo").ok()?;
return text.lines().find(|l| l.starts_with("MemTotal:")).and_then(|l| l.split_whitespace().nth(1)).and_then(|kb| kb.parse::<u64>().ok()).map(|kb| kb / 1024);
}
let out = run_timeout(Command::new("sysctl").args(["-n", "hw.memsize"]), None, Duration::from_secs(5))?;
out.trim().parse::<u64>().ok().map(|b| b / (1024 * 1024))
}

1033
app/igneum-app/src/ember.rs Normal file

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,501 @@
//! Hot-plug: what changed between two enumerations of the cards (src/detect.rs runs one at start and one every
//! minute; the Windows host also asks for one on WM_DEVICECHANGE). Pure, so the rules are unit-tested here; the
//! engine applies the result (start a worker, stop one, mark a row). Born 5 October 2026, when an RX 9070 XT went
//! into PC 1 through an eGPU box while the app ran and nothing noticed.
//!
//! Rules: a card is the same card when its key (vendor:code, "#2" for a twin) matches and the PCI addresses do not
//! disagree, or, failing that, when vendor and name match and that pair is unique on both sides (a twin whose
//! ordinal moved because the first one left). The device index is never part of the identity: it moves. A
//! card missing from a list is removed only when the tool that lists its vendor answered. Removed and faulty cards
//! stay in the engine's list (the other cards' indices are the miner slots), marked, and the dashboard hides a
//! removed row after five minutes.
use crate::config::CardPref;
use crate::state::CardState;
/// How long a removed card's row says "removed" before it hides.
pub const REMOVED_SHOWN_S: f64 = 300.0;
/// How often the engine enumerates again, seconds (macOS has no GPU hot-plug on Apple silicon: slower there).
pub const POLL_S: u64 = if cfg!(target_os = "macos") { 300 } else { 60 };
/// How often the app log carries the full card list, seconds (the console reads it from the log tail).
pub const CARDS_LINE_S: u64 = 600;
#[derive(Default, Debug)]
pub struct Diff {
/// new cards (usable or with a problem), to be appended
pub added: Vec<CardState>,
/// cards that were marked removed earlier and are listed again: (slot, the fresh entry)
pub revived: Vec<(usize, CardState)>,
/// slots whose card is gone
pub removed: Vec<usize>,
/// slots whose card now reports a problem: (slot, "Code 43")
pub errored: Vec<(usize, String)>,
/// slots whose card had a problem and is now driven by a tool again: (slot, the fresh entry)
pub recovered: Vec<(usize, CardState)>,
/// slots whose card is the same but its device index (or memory, bus) changed: (slot, the fresh entry)
pub moved: Vec<(usize, CardState)>,
pub unchanged: Vec<usize>,
}
impl Diff {
/// Nothing to do: every present card is where it was.
pub fn is_quiet(&self) -> bool {
self.added.is_empty() && self.revived.is_empty() && self.removed.is_empty() && self.errored.is_empty() && self.recovered.is_empty() && self.moved.is_empty()
}
}
fn bus_compat(a: &CardState, b: &CardState) -> bool {
a.bus.is_empty() || b.bus.is_empty() || a.bus == b.bus
}
fn same_identity(a: &CardState, b: &CardState) -> bool {
a.vendor == b.vendor && (a.name.trim().eq_ignore_ascii_case(b.name.trim()) || (!a.code.is_empty() && a.code.eq_ignore_ascii_case(&b.code))) && bus_compat(a, b)
}
/// Compares the engine's list with a fresh enumeration. `listed(card)` says whether this enumeration's tools could
/// have seen that card (Detection::listed); a card its tool did not answer for is kept, not removed.
pub fn diff(old: &[CardState], fresh: &[CardState], listed: &dyn Fn(&CardState) -> bool) -> Diff {
let mut out = Diff::default();
let mut used = vec![false; fresh.len()];
let mut pair: Vec<Option<usize>> = vec![None; old.len()];
// exact keys first
for (i, o) in old.iter().enumerate() {
if let Some(j) = fresh.iter().enumerate().position(|(j, f)| !used[j] && f.key == o.key && bus_compat(o, f)) {
used[j] = true;
pair[i] = Some(j);
}
}
// then vendor + name, when that pair is unique among what is still unmatched on both sides
for (i, o) in old.iter().enumerate() {
if pair[i].is_some() {
continue;
}
let cands: Vec<usize> = fresh.iter().enumerate().filter(|(j, f)| !used[*j] && same_identity(o, f)).map(|(j, _)| j).collect();
let twins = old.iter().enumerate().filter(|(k, x)| pair[*k].is_none() && *k != i && same_identity(o, x)).count();
if cands.len() == 1 && twins == 0 {
used[cands[0]] = true;
pair[i] = Some(cands[0]);
}
}
for (i, o) in old.iter().enumerate() {
match pair[i] {
Some(j) => {
let f = &fresh[j];
if o.removed_at > 0.0 {
out.revived.push((i, f.clone()));
} else if o.problem.is_empty() && !f.problem.is_empty() {
out.errored.push((i, f.problem.clone()));
} else if !o.problem.is_empty() && f.problem.is_empty() {
out.recovered.push((i, f.clone()));
} else if !o.problem.is_empty() {
// still faulty: the same problem or a new code, nothing to start or stop
if o.problem != f.problem {
out.errored.push((i, f.problem.clone()));
} else {
out.unchanged.push(i);
}
} else if o.device != f.device || o.key != f.key {
out.moved.push((i, f.clone()));
} else {
out.unchanged.push(i);
}
}
None => {
if o.removed_at > 0.0 {
// already removed: stays hidden or shown as removed
out.unchanged.push(i);
} else if listed(o) {
out.removed.push(i);
} else {
out.unchanged.push(i);
}
}
}
}
for (j, f) in fresh.iter().enumerate() {
if !used[j] {
out.added.push(f.clone());
}
}
out
}
/// The saved choice for a card: by its key (vendor:code), else a key saved by an app before 0.3.11 (vendor:index:code,
/// vendor:index:name) when exactly one matches; an index that moved never changes the answer.
pub fn pref_for<'a>(prefs: &'a std::collections::HashMap<String, CardPref>, c: &CardState) -> Option<&'a CardPref> {
if let Some(p) = prefs.get(&c.key) {
return Some(p);
}
if c.key.contains('#') {
return None; // a twin's choice is its own
}
let head = format!("{}:", c.vendor);
for tail in [format!(":{}", c.code), format!(":{}", c.name)] {
if tail.len() <= 1 {
continue;
}
let found: Vec<&CardPref> = prefs.iter().filter(|(k, _)| k.starts_with(&head) && k.ends_with(&tail) && !k.contains('#')).map(|(_, p)| p).collect();
if found.len() == 1 {
return Some(found[0]);
}
}
None
}
/// Applies a saved choice to a freshly detected card (the first detection and every later one use this).
pub fn apply_pref(c: &mut CardState, p: &CardPref) {
if !c.problem.is_empty() {
return;
}
c.enabled = p.enabled && c.kind != "unknown";
c.identities = p.identities.clamp(1, 64);
if c.enabled || c.kind != "integrated" {
c.reason = String::new();
}
if c.vendor == "nvidia" && p.power_pct > 0 {
c.power_pct = p.power_pct.clamp(crate::sweep::MIN_PCT, 100);
}
c.pinned = p.pinned;
c.sweep_pct = p.sweep_pct;
c.sweep_eff = p.sweep_eff;
c.sweep_watts = p.sweep_watts;
c.sweep_mhs = p.sweep_mhs;
c.sweep_at = p.sweep_at as f64;
// Ember Tune (src/ember.rs): the clock cap the last tune chose, its plan, and the row's Tuned line
c.tune_clock_mhz = p.sweep_clock_mhz;
c.clock_cap_mhz = if p.pinned || p.sweep_source == "baseline" { 0 } else { p.sweep_clock_mhz };
c.tune_source = p.sweep_source.clone();
if p.sweep_mhs > 0.0 && p.sweep_watts > 0.0 {
c.tune_line = crate::ember::tuned_line(p.sweep_mhs, p.sweep_watts, p.sweep_eff);
}
}
/// A card a re-detection added (or brought back): the saved choice if there is one, else the detect defaults it
/// came with; its slot and the time it appeared.
pub fn settle_new(c: &mut CardState, index: usize, pref: Option<&CardPref>, now: f64) {
c.index = index;
c.added_at = now;
c.removed_at = 0.0;
c.gone = false;
if let Some(p) = pref {
apply_pref(c, p);
}
if c.problem.is_empty() {
c.state = if c.enabled { "waiting".into() } else { "off".into() };
}
}
/// Marks a card unplugged: no worker, the row says removed, the saved choice is untouched.
pub fn mark_removed(c: &mut CardState, now: f64) {
c.removed_at = now;
c.gone = false;
c.state = "removed".into();
c.message = "removed".into();
c.hash_now = 0.0;
c.pid = 0;
c.restart_in_s = 0;
c.ready = false;
c.prepared = false;
}
/// A removed row hides after REMOVED_SHOWN_S; returns true when something changed.
pub fn age(cards: &mut [CardState], now: f64) -> bool {
let mut changed = false;
for c in cards.iter_mut() {
let gone = c.removed_at > 0.0 && now - c.removed_at >= REMOVED_SHOWN_S;
if gone != c.gone {
c.gone = gone;
changed = true;
}
}
changed
}
/// The event line for a card that appeared: "New card: <name>, mining" and its kind (ok | info | warn).
pub fn added_words(c: &CardState) -> (&'static str, String) {
if !c.problem.is_empty() {
("warn", format!("New card: {}, not usable ({}); {}", c.name, c.problem, crate::detect::PROBLEM_HINT))
} else if c.enabled {
("ok", format!("New card: {}, mining", c.name))
} else if c.kind == "integrated" {
("info", format!("New card: {}, off ({})", c.name, crate::detect::INTEGRATED_REASON))
} else {
("info", format!("New card: {}, off (switched off in settings)", c.name))
}
}
/// One word for a card's state on the log line and the console: mining, waiting, off, removed, not usable (Code 43).
pub fn state_word(c: &CardState) -> String {
if c.removed_at > 0.0 {
"removed".into()
} else if !c.problem.is_empty() {
format!("not usable ({})", c.problem)
} else if !c.enabled {
"off".into()
} else {
c.state.clone()
}
}
/// The app-log line the console reads (relay/lib/parse.mjs): `cards: <name> [<kind>, <state>] | ...`, hidden rows
/// left out, `cards: none` when nothing is listed.
pub fn cards_line(cards: &[CardState]) -> String {
let parts: Vec<String> = cards.iter().filter(|c| !c.gone).map(|c| format!("{} [{}, {}]", c.name.replace('|', "/").replace('[', "(").replace(']', ")"), c.kind, state_word(c))).collect();
if parts.is_empty() { "cards: none".into() } else { format!("cards: {}", parts.join(" | ")) }
}
#[cfg(test)]
mod tests {
use super::*;
use crate::detect::{classify_kind, mark_unusable, INTEGRATED_REASON};
fn card(name: &str, vendor: &str, device: &str) -> CardState {
let mut c = CardState { index: 0, key: format!("{vendor}:{name}"), code: name.into(), name: name.into(), vendor: vendor.into(), worker: if vendor == "nvidia" { "CUDA".into() } else { "OpenCL".into() }, device: device.into(), enabled: true, state: "off".into(), ..Default::default() };
c.kind = classify_kind(name, None).into();
crate::detect::apply_defaults(&mut c);
c
}
// PC 1 at start on 5 October 2026: the 5090 on nvidia-smi index 0, the Ryzen iGPU as OpenCL device 0 (gfx1036)
fn pc1_start() -> Vec<CardState> {
let mut a = card("NVIDIA GeForce RTX 5090", "nvidia", "0");
a.bus = "00000000:01:00.0".into();
a.state = "mining".into();
let mut b = card("gfx1036", "amd", "0");
b.index = 1;
vec![a, b]
}
fn all_listed(_: &CardState) -> bool {
true
}
fn fresh_pc1_with_egpu() -> Vec<CardState> {
let mut v = pc1_start();
v[0].state = "off".into();
let mut e = card("gfx1201", "amd", "1");
e.index = 2;
v.push(e);
v
}
#[test]
fn unchanged_list_is_quiet() {
let old = pc1_start();
let mut fresh = pc1_start();
fresh[0].state = "off".into(); // runtime state in the fresh list means nothing
let d = diff(&old, &fresh, &all_listed);
assert!(d.is_quiet(), "{d:?}");
assert_eq!(d.unchanged, vec![0, 1]);
}
#[test]
fn a_new_usable_card_is_added_and_nothing_else_moves() {
let old = pc1_start();
let d = diff(&old, &fresh_pc1_with_egpu(), &all_listed);
assert_eq!(d.added.len(), 1);
assert_eq!(d.added[0].name, "gfx1201");
assert_eq!(d.added[0].kind, "discrete");
assert!(d.added[0].enabled);
assert_eq!(d.unchanged, vec![0, 1]);
assert!(d.removed.is_empty() && d.moved.is_empty() && d.errored.is_empty());
}
#[test]
fn a_new_card_with_a_problem_is_added_as_unusable() {
let old = pc1_start();
let mut fresh = pc1_start();
let mut bad = card("AMD Radeon RX 9070 XT", "amd", "");
mark_unusable(&mut bad, "Code 43");
fresh.push(bad);
let d = diff(&old, &fresh, &all_listed);
assert_eq!(d.added.len(), 1);
assert_eq!(d.added[0].problem, "Code 43");
assert!(!d.added[0].enabled);
let (kind, text) = added_words(&d.added[0]);
assert_eq!(kind, "warn");
assert!(text.starts_with("New card: AMD Radeon RX 9070 XT, not usable (Code 43); reboot with the card attached"), "{text}");
assert_eq!(state_word(&d.added[0]), "not usable (Code 43)");
}
#[test]
fn an_unplugged_card_is_removed_only_when_its_tool_answered() {
let old = fresh_pc1_with_egpu();
let fresh = pc1_start();
let d = diff(&old, &fresh, &all_listed);
assert_eq!(d.removed, vec![2]);
assert_eq!(d.unchanged, vec![0, 1]);
// the OpenCL worker did not answer this round: nothing is called removed
let opencl_dead = |c: &CardState| c.vendor == "nvidia";
let d2 = diff(&old, &pc1_start().into_iter().filter(|c| c.vendor == "nvidia").collect::<Vec<_>>(), &opencl_dead);
assert!(d2.removed.is_empty(), "{d2:?}");
assert!(d2.is_quiet());
}
#[test]
fn a_card_whose_index_moved_keeps_its_slot() {
// the eGPU landed before the iGPU in the OpenCL list: the iGPU is device 1 now, the eGPU device 0
let old = pc1_start();
let mut fresh = pc1_start();
fresh[1].device = "1".into();
let mut e = card("gfx1201", "amd", "0");
e.index = 2;
fresh.push(e);
let d = diff(&old, &fresh, &all_listed);
assert_eq!(d.moved.len(), 1);
assert_eq!(d.moved[0].0, 1);
assert_eq!(d.moved[0].1.device, "1");
assert_eq!(d.added.len(), 1);
assert!(d.removed.is_empty());
}
#[test]
fn two_identical_cards_are_told_apart_by_key_and_never_swapped() {
let mut a = card("NVIDIA GeForce RTX 5090", "nvidia", "0");
let mut b = card("NVIDIA GeForce RTX 5090", "nvidia", "1");
b.index = 1;
b.key = "nvidia:NVIDIA GeForce RTX 5090#2".into();
a.bus = "01:00.0".into();
b.bus = "02:00.0".into();
let old = vec![a.clone(), b.clone()];
// the second twin leaves: the first keeps its slot by key; the missing one is removed, not "moved"
let d = diff(&old, &[a.clone()], &all_listed);
assert_eq!(d.unchanged, vec![0]);
assert_eq!(d.removed, vec![1]);
// the FIRST twin leaves: the survivor is now index 0 with the unsuffixed key, but its bus says which card it
// is, so slot 1 is "moved" (new key and device) and slot 0 is removed; no worker is swapped between cards
let mut survivor = b.clone();
survivor.device = "0".into();
survivor.key = "nvidia:NVIDIA GeForce RTX 5090".into();
let d2 = diff(&old, &[survivor], &all_listed);
assert_eq!(d2.removed, vec![0]);
assert_eq!(d2.moved.len(), 1);
assert_eq!(d2.moved[0].0, 1);
assert_eq!(d2.moved[0].1.key, "nvidia:NVIDIA GeForce RTX 5090");
// the same two cards again, nothing changed: quiet
let d3 = diff(&old, &old, &all_listed);
assert!(d3.is_quiet(), "{d3:?}");
}
#[test]
fn a_driven_card_that_turns_faulty_is_errored_and_recovers_later() {
let old = fresh_pc1_with_egpu();
// the eGPU is still listed by Windows, now with Code 43, and no longer by OpenCL
let mut fresh = pc1_start();
let mut bad = card("gfx1201", "amd", "");
mark_unusable(&mut bad, "Code 43");
fresh.push(bad);
let d = diff(&old, &fresh, &all_listed);
assert_eq!(d.errored, vec![(2, "Code 43".to_string())]);
assert!(d.added.is_empty() && d.removed.is_empty());
// after a reboot with the card attached it is driven again: recovered, same slot
let mut faulty = old.clone();
mark_unusable(&mut faulty[2], "Code 43");
faulty[2].device = String::new();
let d2 = diff(&faulty, &fresh_pc1_with_egpu(), &all_listed);
assert_eq!(d2.recovered.len(), 1);
assert_eq!(d2.recovered[0].0, 2);
assert!(d2.recovered[0].1.problem.is_empty());
// the same problem again next minute: quiet
let d3 = diff(&faulty, &fresh, &all_listed);
assert!(d3.is_quiet(), "{d3:?}");
}
#[test]
fn a_removed_card_that_comes_back_is_revived_in_its_slot() {
let mut old = fresh_pc1_with_egpu();
mark_removed(&mut old[2], 1000.0);
assert_eq!(state_word(&old[2]), "removed");
// still absent: quiet (and hidden after five minutes)
let d = diff(&old, &pc1_start(), &all_listed);
assert!(d.is_quiet(), "{d:?}");
assert!(age(&mut old, 1000.0 + REMOVED_SHOWN_S));
assert!(old[2].gone);
assert!(!age(&mut old, 1000.0 + REMOVED_SHOWN_S + 1.0));
// back: revived in slot 2
let d2 = diff(&old, &fresh_pc1_with_egpu(), &all_listed);
assert_eq!(d2.revived.len(), 1);
assert_eq!(d2.revived[0].0, 2);
assert!(d2.added.is_empty());
let mut back = d2.revived[0].1.clone();
settle_new(&mut back, 2, None, 2000.0);
assert_eq!(back.index, 2);
assert_eq!(back.removed_at, 0.0);
assert!(!back.gone);
assert_eq!(back.added_at, 2000.0);
assert_eq!(back.state, "waiting");
}
#[test]
fn the_users_choice_is_kept_on_a_new_card_and_an_integrated_one_is_off_by_default() {
let mut prefs = std::collections::HashMap::new();
// the user switched the iGPU on earlier and set 2 identities; the setting was saved under OpenCL index 0
prefs.insert("amd:0:gfx1036".to_string(), CardPref { enabled: true, identities: 2, ..Default::default() });
// the iGPU comes back as device 1 (the eGPU took index 0): the 0.3.9 key still answers, by vendor and code
let mut igpu = card("gfx1036", "amd", "1");
assert_eq!(igpu.kind, "integrated");
assert!(!igpu.enabled);
assert_eq!(igpu.reason, INTEGRATED_REASON);
let p = pref_for(&prefs, &igpu).cloned();
assert!(p.is_some());
settle_new(&mut igpu, 1, p.as_ref(), 5.0);
assert!(igpu.enabled);
assert_eq!(igpu.identities, 2);
assert_eq!(igpu.reason, "");
assert_eq!(igpu.state, "waiting");
// the user switched it off (saved under the index-free key): the row keeps the integrated words
prefs.insert("amd:gfx1036".to_string(), CardPref { enabled: false, identities: 1, ..Default::default() });
let mut igpu2 = card("gfx1036", "amd", "1");
let p2 = pref_for(&prefs, &igpu2).cloned();
settle_new(&mut igpu2, 1, p2.as_ref(), 6.0);
assert!(!igpu2.enabled);
assert_eq!(igpu2.reason, INTEGRATED_REASON);
assert_eq!(igpu2.state, "off");
let (kind, text) = added_words(&igpu2);
assert_eq!(kind, "info");
assert_eq!(text, format!("New card: gfx1036, off ({INTEGRATED_REASON})"));
// no saved choice: the detect default (integrated off, discrete on)
let mut egpu = card("gfx1201", "amd", "0");
settle_new(&mut egpu, 2, None, 7.0);
assert!(egpu.enabled);
assert_eq!(added_words(&egpu), ("ok", "New card: gfx1201, mining".to_string()));
// a pref never switches on a card with a problem
let mut bad = card("AMD Radeon RX 9070 XT", "amd", "");
mark_unusable(&mut bad, "Code 43");
settle_new(&mut bad, 3, Some(&CardPref { enabled: true, identities: 8, ..Default::default() }), 8.0);
assert!(!bad.enabled);
assert_eq!(bad.state, "unusable");
// old keys only, two of them for the same code (the five-row PC 1 list had amd:0:gfx1036 and amd:2:gfx1036):
// ambiguous, so the default applies; the index-free key, once saved, always wins
prefs.remove("amd:gfx1036");
prefs.insert("amd:2:gfx1036".to_string(), CardPref { enabled: true, identities: 3, ..Default::default() });
let other = card("gfx1036", "amd", "7");
assert!(pref_for(&prefs, &other).is_none());
prefs.insert("amd:gfx1036".to_string(), CardPref { enabled: true, identities: 4, ..Default::default() });
assert_eq!(pref_for(&prefs, &other).map(|p| p.identities), Some(4));
// a twin never borrows the first card's choice
let mut twin = card("gfx1201", "amd", "3");
twin.key = "amd:gfx1201#2".into();
prefs.insert("amd:gfx1201".to_string(), CardPref { enabled: false, identities: 1, ..Default::default() });
assert!(pref_for(&prefs, &twin).is_none());
// the name on the row is the Windows name while the key keeps the code: the 0.3.9 key by code still answers
let mut named = card("gfx1201", "amd", "1");
named.name = "AMD Radeon RX 9070 XT".into();
prefs.clear();
prefs.insert("amd:1:gfx1201".to_string(), CardPref { enabled: false, identities: 2, ..Default::default() });
assert_eq!(pref_for(&prefs, &named).map(|p| p.identities), Some(2));
}
#[test]
fn the_console_line_lists_every_shown_card_with_kind_and_state() {
let mut cards = fresh_pc1_with_egpu();
cards[0].state = "mining".into();
cards[2].state = "starting".into();
let mut bad = card("AMD Radeon RX 9070 XT", "amd", "");
mark_unusable(&mut bad, "Code 43");
cards.push(bad);
assert_eq!(cards_line(&cards), "cards: NVIDIA GeForce RTX 5090 [discrete, mining] | gfx1036 [integrated, off] | gfx1201 [discrete, starting] | AMD Radeon RX 9070 XT [discrete, not usable (Code 43)]");
mark_removed(&mut cards[2], 10.0);
assert!(cards_line(&cards).contains("gfx1201 [discrete, removed]"));
age(&mut cards, 10.0 + REMOVED_SHOWN_S);
assert!(!cards_line(&cards).contains("gfx1201"));
assert_eq!(cards_line(&[]), "cards: none");
}
}

View file

@ -1115,13 +1115,11 @@ fn run_script(shared: &Arc<Shared>, job: &Job, sink: &Sink, jobs_dir: &Path, dat
.map(|(k, v)| format!("$env:{k} = '{}'\r\n", v.replace('\'', "''")))
.collect();
let wrapper = dir.join("elevated.ps1");
let w = format!("{env_lines}& '{}' *>&1 | Out-File -FilePath '{}' -Encoding utf8\r\nexit $LASTEXITCODE\r\n", script.display().to_string().replace('\'', "''"), out_file.display().to_string().replace('\'', "''"));
let w = elevated_wrapper(&env_lines, &script.display().to_string(), &out_file.display().to_string());
std::fs::write(&wrapper, [b"\xEF\xBB\xBF".as_slice(), w.as_bytes()].concat()).map_err(|e| e.to_string())?;
let _ = std::fs::remove_file(&out_file);
let inner = format!("-NoProfile -ExecutionPolicy Bypass -File \"{}\"", wrapper.display());
let ps = format!("$p = Start-Process -FilePath powershell.exe -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru; exit $p.ExitCode", inner.replace('\'', "''"));
cmd = Command::new(crate::platform::tool("powershell"));
cmd.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &ps]);
cmd = crate::platform::elevated_command("powershell.exe", &inner);
} else if shell == "powershell" {
cmd = Command::new(crate::platform::tool("powershell"));
cmd.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-File", &script.display().to_string()]);
@ -1131,10 +1129,17 @@ fn run_script(shared: &Arc<Shared>, job: &Job, sink: &Sink, jobs_dir: &Path, dat
}
cmd.current_dir(&dir);
job_env(&mut cmd, shared, job, &dir, data_root);
// the elevated script's output reaches this side through a file: follow it while the script runs, so the
// 5-minute progress reports carry its lines (6 October 2026: a 35-minute run that never mined showed only
// "script running" until it ended; the lines that said why were in the file the whole time)
let follow = if elevated { Some(follow_file(sink, out_file.clone())) } else { None };
let ran = run_streamed(&mut cmd, sink, ctl, limit, shared, job, started, "script running")?;
if elevated {
if let Some(f) = follow {
f.stop.store(true, std::sync::atomic::Ordering::Relaxed);
let seen = f.handle.join().unwrap_or(0);
// the tail the follower had not read when the script ended
if let Ok(t) = std::fs::read_to_string(&out_file) {
for l in t.lines() {
for l in t.lines().skip(seen) {
sink.line(l);
}
}
@ -1142,9 +1147,60 @@ fn run_script(shared: &Arc<Shared>, job: &Job, sink: &Sink, jobs_dir: &Path, dat
finish_ran(ran, "script")
}
/// Follows a file another process writes (the elevated script's output), feeding each new complete line to the
/// sink every 2 s until stopped; returns how many lines it delivered, so the caller can hand over the remainder.
struct Follow {
stop: Arc<std::sync::atomic::AtomicBool>,
handle: std::thread::JoinHandle<usize>,
}
fn follow_file(sink: &Sink, path: PathBuf) -> Follow {
let stop = Arc::new(std::sync::atomic::AtomicBool::new(false));
let stop2 = stop.clone();
let s = Sink { shared: sink.shared.clone(), id: sink.id.clone(), dir: sink.dir.clone(), log_path: sink.log_path.clone(), file: Mutex::new(std::fs::OpenOptions::new().append(true).open(&sink.log_path).ok()), results: Mutex::new(Vec::new()) };
let handle = std::thread::spawn(move || {
let mut seen = 0usize;
loop {
if let Ok(t) = std::fs::read_to_string(&path) {
let lines: Vec<&str> = t.lines().collect();
// only complete lines (the writer may be mid-line): keep the last one for the next pass unless the
// text ends with a newline
let complete = if t.ends_with('\n') { lines.len() } else { lines.len().saturating_sub(1) };
for l in lines.iter().take(complete).skip(seen) {
s.line(l);
}
seen = seen.max(complete);
}
if stop2.load(std::sync::atomic::Ordering::Relaxed) {
break seen;
}
std::thread::sleep(Duration::from_secs(2));
}
});
Follow { stop, handle }
}
/// The PowerShell wrapper an elevated job runs (its own process, its own environment): the IGNEUM_* values, then one
/// line about its console (the elevated process cannot inherit the engine's headless console and gets one of its own;
/// `-WindowStyle Hidden` on the launch keeps it hidden, and this line is the running measurement of that on every
/// elevated job: "elevated console: hwnd N visible False"), then the script, everything into `out_file` for the engine
/// to read back. The console-window class, PC 1, 5 October 2026 (tools/windows/console-watch-elevated.ps1).
fn elevated_wrapper(env_lines: &str, script: &str, out_file: &str) -> String {
let (script, out) = (crate::platform::ps_quote(script), crate::platform::ps_quote(out_file));
format!(
"{env_lines}$ErrorActionPreference = 'Continue'\r\n\
$igc = ''\r\n\
try {{ Add-Type -Name IgCon -Namespace Igneum -MemberDefinition '[DllImport(\"kernel32.dll\")] public static extern System.IntPtr GetConsoleWindow(); [DllImport(\"user32.dll\")] public static extern bool IsWindowVisible(System.IntPtr h);'; $h = [Igneum.IgCon]::GetConsoleWindow(); $igc = \"elevated console: hwnd $h visible $([Igneum.IgCon]::IsWindowVisible($h))\" }} catch {{ $igc = \"elevated console: unknown ($_)\" }}\r\n\
$igc | Out-File -FilePath '{out}' -Encoding utf8\r\n\
& '{script}' *>&1 | Out-File -FilePath '{out}' -Encoding utf8 -Append\r\n\
exit $LASTEXITCODE\r\n"
)
}
fn finish_ran(ran: Ran, what: &str) -> Result<Done, String> {
match ran.code {
Some(0) => Ok(Done { status: "done".into(), exit: 0, summary: format!("{what} finished, exit 0"), extra: json!({}) }),
Some(251) => Ok(Done { status: "failed".into(), exit: 251, summary: format!("{what} did not start: the administrator prompt was refused, cancelled or timed out (click Yes within 2 minutes)"), extra: json!({}) }),
Some(c) => Ok(Done { status: "failed".into(), exit: c as i64, summary: format!("{what} exited with code {c}"), extra: json!({}) }),
None if ran.timed_out => Ok(Done { status: "timeout".into(), exit: -1, summary: format!("{what} hit the time cap and was ended"), extra: json!({}) }),
None => Err(format!("{what} was ended")),
@ -1259,6 +1315,34 @@ fn collect_done(uploaded: u32, failed: u32, names: Vec<String>, ran: Option<Ran>
mod tests {
use super::*;
#[test]
fn a_refused_administrator_prompt_is_a_failure_not_done() {
// 5 October 2026: the elevated launcher exited 0 after Windows cancelled an unanswered UAC prompt
let d = finish_ran(Ran { code: Some(251), timed_out: false }, "script").unwrap();
assert_eq!((d.status.as_str(), d.exit), ("failed", 251));
assert!(d.summary.contains("administrator prompt"), "{}", d.summary);
let d = finish_ran(Ran { code: Some(0), timed_out: false }, "script").unwrap();
assert_eq!(d.status, "done");
// the launcher string itself (platform::elevated_ps_line since 13755b9): a thrown Start-Process must not fall
// through to `exit $p.ExitCode`
let l = crate::platform::elevated_ps_line("powershell.exe", "-NoProfile -File x.ps1");
assert!(l.contains("-Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}");
assert!(l.contains("if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }"), "{l}");
}
#[test]
fn elevated_wrapper_reports_its_console_then_runs_the_script() {
let w = elevated_wrapper("$env:IGNEUM_JOB_ID = 'j1'\r\n", r"C:\jobs\j1\script.ps1", r"C:\jobs\it's\elevated-output.log");
assert!(w.starts_with("$env:IGNEUM_JOB_ID = 'j1'\r\n$ErrorActionPreference = 'Continue'\r\n"), "{w}");
assert!(w.contains("GetConsoleWindow()") && w.contains("IsWindowVisible("), "{w}");
assert!(w.contains("$igc | Out-File -FilePath 'C:\\jobs\\it''s\\elevated-output.log' -Encoding utf8\r\n"), "{w}");
assert!(w.contains("& 'C:\\jobs\\j1\\script.ps1' *>&1 | Out-File -FilePath 'C:\\jobs\\it''s\\elevated-output.log' -Encoding utf8 -Append\r\n"), "{w}");
assert!(w.ends_with("exit $LASTEXITCODE\r\n"), "{w}");
// every line ends in CRLF (the file is written for Windows PowerShell): the env line and six of its own
assert_eq!(w.matches("\r\n").count(), 7, "{w:?}");
assert_eq!(w.matches('\n').count(), 7, "{w:?}");
}
#[test]
fn collect_outcome_follows_the_command_exit() {
let d = collect_done(0, 0, vec![], None);
@ -1743,6 +1827,7 @@ fn spawn_relaunch_helper(shared: &Arc<Shared>) -> Result<(), String> {
{
let dir = std::env::current_exe().ok().and_then(|p| p.parent().map(|d| d.to_path_buf())).ok_or("cannot find the install folder")?;
let exe = dir.join("igneum-app.exe");
// console: igneum-app.exe is a windows-subsystem program in release builds (main.rs), it never gets a console; SW_HIDE would hide the window host it opens
let ps = format!("Start-Sleep 8; Start-Process -FilePath '{}' -ArgumentList '--launch' -WorkingDirectory '{}'", exe.display().to_string().replace('\'', "''"), dir.display().to_string().replace('\'', "''"));
c = Command::new(crate::platform::tool("powershell"));
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-WindowStyle", "Hidden", "-Command", &ps]);

View file

@ -16,6 +16,7 @@
mod config;
mod detect;
mod engine;
mod hotplug;
mod keys;
mod platform;
mod procs;
@ -28,9 +29,12 @@ mod jobs;
mod jobrun;
mod jobbuild;
mod prover;
mod provedefault;
mod segments;
mod verifier;
mod wslhost;
mod sweep;
mod ember;
mod watchdog;
use std::io::{BufRead, Write};
@ -132,16 +136,18 @@ fn main() {
let Ok(l) = line else { break };
let t = l.trim();
match t {
"quit" => shared.send(engine::Cmd::Quit),
"quit" => shared.send(engine::Cmd::Quit("the window host (quit on stdin: the tray menu or the installer)")),
"pause" => shared.send(engine::Cmd::Pause),
"resume" => shared.send(engine::Cmd::Resume),
// the window host saw WM_DEVICECHANGE (a card plugged in or out): enumerate now, not at the next minute
"detect" => shared.send(engine::Cmd::Detect),
"elevated ok" => shared.send(engine::Cmd::ElevatedDone(Ok(()))),
_ if t.starts_with("elevated fail") => shared.send(engine::Cmd::ElevatedDone(Err(t.trim_start_matches("elevated fail").trim_start_matches(':').trim().to_string()))),
_ => {}
}
}
if wrapper {
shared.send(engine::Cmd::Quit);
shared.send(engine::Cmd::Quit("the window host went away (stdin closed)"));
}
});
}

View file

@ -546,3 +546,42 @@ mod tests {
assert_eq!(fingerprint("zz"), "");
}
}
/// Unix seconds of a manifest `published_at` ("2026-10-04T13:00:00Z", whole seconds, UTC); `None` for any other shape.
pub fn unix_from_rfc3339(t: &str) -> Option<u64> {
let t = t.trim();
let b = t.as_bytes();
if b.len() < 20 || b[4] != b'-' || b[7] != b'-' || b[10] != b'T' || b[13] != b':' || b[16] != b':' || !t.ends_with('Z') {
return None;
}
let n = |a: usize, z: usize| t[a..z].parse::<i64>().ok();
let (y, m, d, hh, mm, ss) = (n(0, 4)?, n(5, 7)?, n(8, 10)?, n(11, 13)?, n(14, 16)?, n(17, 19)?);
if !(1..=12).contains(&m) || !(1..=31).contains(&d) || hh > 23 || mm > 59 || ss > 60 {
return None;
}
// days from civil (Howard Hinnant), valid for every date after 1970
let (y2, m2) = if m <= 2 { (y - 1, m + 9) } else { (y, m - 3) };
let era = y2.div_euclid(400);
let yoe = y2 - era * 400;
let doy = (153 * m2 + 2) / 5 + d - 1;
let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy;
let days = era * 146097 + doe - 719468;
if days < 0 {
return None;
}
Some((days as u64) * 86400 + (hh as u64) * 3600 + (mm as u64) * 60 + ss as u64)
}
#[cfg(test)]
mod rfc3339_tests {
use super::unix_from_rfc3339;
#[test]
fn a_manifest_publish_time_parses_to_unix_seconds() {
assert_eq!(unix_from_rfc3339("1970-01-01T00:00:00Z"), Some(0));
assert_eq!(unix_from_rfc3339("2026-10-05T23:57:49Z"), Some(1791244669));
assert_eq!(unix_from_rfc3339("2026-10-04T13:00:00Z"), Some(1791118800));
assert_eq!(unix_from_rfc3339(""), None);
assert_eq!(unix_from_rfc3339("2026-10-05 23:57:49"), None);
assert_eq!(unix_from_rfc3339("2026-13-05T23:57:49Z"), None);
}
}

View file

@ -34,6 +34,8 @@ use std::process::Command;
use std::sync::Arc;
use std::time::{Duration, Instant};
/// A manifest published this long before the engine started is a catch-up: the hourly rollout slot does not apply.
const CATCH_UP_AFTER_S: u64 = 3600;
const HEALTHY_AFTER_S: u64 = 90;
const CHECK_EVERY_S: u64 = 3600;
const RETRY_AFTER_ERROR_S: u64 = 600;
@ -117,6 +119,11 @@ pub struct Updater {
deferred_until: Option<Instant>,
/// this machine's minute of the hour for applying (manifest::slot_minute of the machine id)
slot: u64,
/// When this engine started (unix seconds): an update published more than an hour before it is a catch-up, not a
/// rollout, and skips the hourly slot (the project lead's morning of 6 October 2026: PC 1 came up after the 0.3.11 publish and
/// sat on "installs at the next safe moment" until he pressed Install now).
started_unix: u64,
catch_up_logged: bool,
/// identity counts from /api/live over the last 10 minutes, sampled while an update is ready
live_samples: Vec<(Instant, u64)>,
live_next: Instant,
@ -163,6 +170,8 @@ impl Updater {
apply_launched: None,
deferred_until: None,
slot: manifest::slot_minute(&shared.runtime.id8()),
started_unix: crate::platform::unix_now(),
catch_up_logged: false,
live_samples: Vec::new(),
live_next: now,
live_busy: false,
@ -519,7 +528,12 @@ impl Updater {
}
};
let minute = (crate::platform::unix_now() / 60) % 60;
let slot_ok = minute == self.slot || std::env::var("IGNEUM_APP_UPDATE_NO_SLOT").map(|v| v == "1").unwrap_or(false);
let catch_up = self.manifest.as_ref().and_then(|m| manifest::unix_from_rfc3339(&m.published_at)).map(|p| p + CATCH_UP_AFTER_S <= self.started_unix).unwrap_or(false);
if catch_up && !self.catch_up_logged {
self.catch_up_logged = true;
shared.log(&format!("update: {} was published over an hour before this start, so it installs at the first safe moment (no hourly slot)", self.version()));
}
let slot_ok = minute == self.slot || catch_up || std::env::var("IGNEUM_APP_UPDATE_NO_SLOT").map(|v| v == "1").unwrap_or(false);
let ready_for = self.ready_since.map(|t| now.duration_since(t).as_secs()).unwrap_or(0);
let moment = Moment { node_synced: ctx.node_synced, boundary_eta_s: ctx.boundary_eta_s, miner_busy: ctx.miner_busy, ready_for_s: ready_for, urgent: urgent || self.install_asked, slot_ok, network_drop_pct };
if !self.auto && !urgent && !self.install_asked {
@ -1264,6 +1278,7 @@ function EngineAlive() { return [bool](Get-Process -Id $EnginePid -ErrorAction S
function Relaunch() {
if (EngineAlive) { return }
$exe = Join-Path $InstallDir 'igneum-app.exe'
# console: igneum-app.exe is a windows-subsystem program (no console); -WindowStyle Hidden would hide the window host it opens
if (Test-Path $exe) { Log 'engine gone and nothing installed: starting the old app again'; Start-Process -FilePath $exe -ArgumentList '--launch' -WorkingDirectory $InstallDir | Out-Null }
}
Log "$Mode : engine $EnginePid installer '$Installer' version $Version (the engine keeps mining until the installer runs)"
@ -1278,6 +1293,7 @@ $setupArgs = @('/VERYSILENT', '/SUPPRESSMSGBOXES', '/NORESTART', '/CLOSEAPPLICAT
try {
# no -Verb RunAs: a per-user installer just runs; an administrator installer makes Windows ask, and a declined or
# timed-out prompt comes back here as an exception with the engine still mining
# console: the Inno Setup installer is a GUI program (no console), /VERYSILENT shows nothing
$p = Start-Process -FilePath $Installer -ArgumentList $setupArgs -Wait -PassThru
if ($p.ExitCode -eq 0) {
if ($Mode -eq 'rollback') { Done $false "Igneum Miner $Version did not stay up twice; the previous version was reinstalled" $true $false }

View file

@ -396,10 +396,7 @@ pub fn sync_clock() -> Result<String, String> {
#[cfg(windows)]
{
let cmd = tool("cmd").display().to_string();
let script = format!("Start-Process -FilePath '{cmd}' -ArgumentList '/c net start w32time & w32tm /resync /force' -Verb RunAs -Wait -WindowStyle Hidden");
let mut c = Command::new(tool("powershell"));
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &script]);
quiet(&mut c);
let mut c = elevated_command(&cmd, "/c net start w32time & w32tm /resync /force");
let out = c.output().map_err(|e| e.to_string())?;
if out.status.success() {
Ok("asked Windows Time to resync (w32tm /resync)".into())
@ -425,18 +422,14 @@ pub fn sync_clock() -> Result<String, String> {
pub fn run_elevated(cmdline: &str) -> Result<(), String> {
#[cfg(windows)]
{
let escaped = cmdline.replace('\'', "''");
let cmd = tool("cmd").display().to_string();
let script = format!("$p = Start-Process -FilePath '{cmd}' -ArgumentList '/c {escaped}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru; exit $p.ExitCode");
let mut c = Command::new(tool("powershell"));
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &script]);
quiet(&mut c);
let mut c = elevated_command(&cmd, &format!("/c {cmdline}"));
let out = c.output().map_err(|e| e.to_string())?;
if out.status.success() {
Ok(())
} else {
let err = String::from_utf8_lossy(&out.stderr).trim().to_string();
Err(if err.contains("canceled") || err.contains("cancelled") || err.is_empty() { "the administrator prompt was cancelled".into() } else { err })
Err(elevated_failure(out.status.code(), &err))
}
}
#[cfg(target_os = "linux")]
@ -451,6 +444,52 @@ pub fn run_elevated(cmdline: &str) -> Result<(), String> {
}
}
/// The reason an elevated step failed, from the launcher's exit code and stderr: exit 251 (the prompt refused,
/// cancelled or timed out, `elevated_ps_line`) and the "canceled" wording name the prompt; any other code is the
/// step's own exit (the engine then keeps Power control on: rights were given).
pub fn elevated_failure(code: Option<i32>, stderr: &str) -> String {
if code == Some(ELEVATED_LAUNCH_FAILED) || stderr.contains("canceled") || stderr.contains("cancelled") {
"the administrator prompt was refused, cancelled or timed out".into()
} else if stderr.is_empty() {
format!("the elevated step exited with code {}", code.map(|c| c.to_string()).unwrap_or_else(|| "?".into()))
} else {
stderr.to_string()
}
}
/// Doubles the single quotes of `s` for a single-quoted PowerShell literal.
pub fn ps_quote(s: &str) -> String {
s.replace('\'', "''")
}
/// The PowerShell line that starts `file args` as administrator (one UAC prompt), waits, and exits with the child's
/// code. Every elevated launch of the app goes through here (the NVIDIA power cap, the sweep helper, the clock sync,
/// an elevated remote job) so the console flags live in one place: `-WindowStyle Hidden` is SW_HIDE on the new
/// process the AppInfo service creates; the elevated child cannot inherit this process's headless console, so without
/// it the child gets a console of its own (5 October 2026, PC 1 watcher, tools/windows/console-watch*.ps1).
/// A refused, cancelled or unanswered prompt makes Start-Process throw and `$p` stay null: that is exit 251 with the
/// reason on stderr, never `exit $p.ExitCode` = 0 (the 5 October 2026 driver job on PC 1 was reported done after
/// Windows cancelled its prompt at 122 s).
pub fn elevated_ps_line(file: &str, args: &str) -> String {
format!(
"try {{ $p = Start-Process -FilePath '{}' -ArgumentList '{}' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop }} catch {{ Write-Error ('elevated launch failed (UAC refused, cancelled or timed out): ' + $_.Exception.Message); exit 251 }}; if ($null -eq $p) {{ Write-Error 'elevated launch failed: no process'; exit 251 }}; exit $p.ExitCode",
ps_quote(file),
ps_quote(args)
)
}
/// The exit code `elevated_ps_line` uses when the elevated process never started (the prompt refused, cancelled or
/// timed out).
pub const ELEVATED_LAUNCH_FAILED: i32 = 251;
/// The hidden PowerShell that runs `elevated_ps_line(file, args)`: blocking when run, one UAC prompt on the PC.
pub fn elevated_command(file: &str, args: &str) -> Command {
let mut c = Command::new(tool("powershell"));
c.args(["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command", &elevated_ps_line(file, args)]);
quiet(&mut c);
c
}
/// Builds a command that runs without a console window on Windows.
pub fn quiet(cmd: &mut Command) -> &mut Command {
#[cfg(windows)]
@ -463,6 +502,34 @@ pub fn quiet(cmd: &mut Command) -> &mut Command {
#[cfg(test)]
mod tests {
#[test]
fn elevated_line_is_hidden_and_quoted() {
let l = super::elevated_ps_line(r"C:\WINDOWS\system32\cmd.exe", "/c echo it's & exit 3");
assert!(l.starts_with("try { $p = "), "{l}");
assert!(l.contains("-FilePath 'C:\\WINDOWS\\system32\\cmd.exe' -ArgumentList '/c echo it''s & exit 3' -Verb RunAs -Wait -WindowStyle Hidden -PassThru -ErrorAction Stop } catch {"), "{l}");
assert!(l.contains("-Verb RunAs"), "{l}");
assert!(l.contains("-WindowStyle Hidden"), "{l}");
// a thrown Start-Process (the prompt refused) never falls through to `exit $p.ExitCode`
assert!(l.contains("exit 251 }; if ($null -eq $p) { Write-Error 'elevated launch failed: no process'; exit 251 }; exit $p.ExitCode"), "{l}");
assert!(l.ends_with("exit $p.ExitCode"), "{l}");
assert_eq!(super::ELEVATED_LAUNCH_FAILED, 251);
assert_eq!(super::elevated_failure(Some(251), "elevated launch failed (UAC refused, cancelled or timed out): ..."), "the administrator prompt was refused, cancelled or timed out");
assert_eq!(super::elevated_failure(Some(1), "The operation was canceled by the user."), "the administrator prompt was refused, cancelled or timed out");
assert_eq!(super::elevated_failure(Some(2), ""), "the elevated step exited with code 2");
assert_eq!(super::elevated_failure(Some(3), "nvidia-smi: bad"), "nvidia-smi: bad");
assert_eq!(super::ps_quote("a'b''c"), "a''b''''c");
assert_eq!(super::ps_quote("plain"), "plain");
}
#[test]
fn elevated_command_is_a_hidden_powershell() {
let c = super::elevated_command("powershell.exe", "-NoProfile -File \"C:\\x y\\elevated.ps1\"");
let args: Vec<String> = c.get_args().map(|a| a.to_string_lossy().into_owned()).collect();
assert_eq!(&args[..4], ["-NoProfile", "-ExecutionPolicy", "Bypass", "-Command"]);
assert!(args[4].contains("-ArgumentList '-NoProfile -File \"C:\\x y\\elevated.ps1\"' -Verb RunAs -Wait -WindowStyle Hidden"), "{}", args[4]);
assert!(c.get_program().to_string_lossy().contains("powershell"));
}
#[test]
fn token_redaction() {
let l = "dashboard at http://127.0.0.1:58776/t/a3a01c537130bceeaa1f6118ba48d63e/ (log x)";

View file

@ -13,6 +13,7 @@ pub enum Source {
Watch,
Miner(usize), // card index
Telemetry, // nvidia-smi -l 5
AmdTelemetry, // igneum-gpu-telemetry -l 5 (ADLX or sysfs), 5 October 2026
}
impl Source {
@ -22,6 +23,7 @@ impl Source {
Source::Watch => "watch".into(),
Source::Miner(i) => format!("miner{}", i + 1),
Source::Telemetry => "gpu".into(),
Source::AmdTelemetry => "gpu-amd".into(),
}
}
}

View file

@ -0,0 +1,169 @@
//! Proving v1 step 1 (5 October 2026, the project lead: "open the proving round asap"): the prover is on by default on every
//! mining machine that can prove, decided once per install after the cards are detected (src/engine.rs
//! `apply_prove_default`). The rule, one line each:
//!
//! | Machine | Default | Why (bench-log 5 October 2026, "proving v1", the S_p curve on the RTX 5090, SP1 6.8.1's GPU prover) |
//! |---|---|---|
//! | NVIDIA card with 24 GB or more, mining or not, Windows with WSL2 (Ubuntu-24.04) answering or Linux | on | a full shard at the adopted v1 budget (30,000 pgas, 4.7 M cycles) peaks at 20,434 MiB alone and 22,210 beside the miner (measured on the 5090; approximate for a 24 GB card's own allocation); the prototype shard the devnet proves until its fee switch (6.75 M pgas) peaks at 28,307 MiB alone and 30,039 beside the miner, so until the switch only a 32 GB card proves it and a 24 GB card's prover waits for shards it can hold (the host refuses nothing; a proof that runs out of memory fails and the shard is left) |
//! | NVIDIA card of 16 to 24 GB | off, with the line saying why | the GPU prover's floor is 13,874 MiB for an EMPTY shard, 15,670 beside the miner; a 16 GB card holds no full shard |
//! | NVIDIA card under 16 GB | off | 13,874 MiB does not fit; the project lead's 12 GB requirement is open until a prover build with a smaller floor is measured |
//! | Windows under 32 GB of RAM | off, with the line saying why | the WSL2 prover held 7.9 GB on a 63 GB PC; a 16 GB PC would swap |
//! | Windows with a qualifying card but WSL2 silent | off, with the Set up hint | nothing can prove until the distribution exists |
//! | Apple silicon | off | the M5 Max CPU took 41 to 55 s for an EMPTY shard's compressed proof under load and 272 s for a 200-pgas shard; a full shard was never under 60 s (bench-log 4 and 5 October 2026) |
//! | AMD-only (no NVIDIA card) | off, "mines and does not prove" | no zkVM proves on an AMD GPU today (docs/analysis/amd-proving.md); the SP1 CPU prover on PC 1 cost 82 to 87 s core plus 199 to 202 s compressed a shard at a 30 GB RSS whatever the shard size (bench-log, "the SP1 CPU prover on PC 1") |
//!
//! Decided 5 October 2026 (delegated by the project lead: "deploy what is absolute best"), docs/plans/proving-v1.md. The default
//! never switches an explicit on back off, and Settings always wins afterwards.
use crate::state::CardState;
/// A card that proves, mining or not, needs this much: the adopted v1 shard peaks at 20,434 MiB alone (the S_p curve,
/// 5 October 2026) and 22,210 beside the miner; `nvidia-smi` reports MiB and a 24 GB card reports 24,564, so the
/// test is at 23 GB. (The same value for a mining and an idle card: the floor is the GPU server's, not the miner's.)
pub const MIN_VRAM_MB_MINING: u64 = 23_552;
pub const MIN_VRAM_MB_PROVE_ONLY: u64 = 23_552;
/// What a 32 GB card alone can do that a 24 GB one cannot: the prototype shard (28,307 MiB alone, 30,039 beside the
/// miner), the devnet's shard until its fee switch at DAA 210,000; `nvidia-smi` reports 32,607 for the RTX 5090.
pub const VRAM_MB_PROTOTYPE_SHARD: u64 = 31_000;
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct Decision {
pub on: bool,
/// One plain sentence for the log and the Proving tile.
pub line: String,
}
fn gb(mb: u64) -> u64 {
(mb + 512) / 1024
}
/// Windows machines under this much RAM stay off until measured (consequences review C4, 5 October 2026): PC 2 at
/// 63 GB had 25.6 GB in use with the WSL2 VM's working set at 7.9 GB while proving; a 16 GB PC would swap.
pub const MIN_RAM_MB_WINDOWS: u64 = 31_000;
/// The card an aggregation (the chained SP1 recursion, spec 7.8) may run on: 16,751 MiB measured with the miner
/// resident (13.4 GB alone, approximate), so a 24 GB card mining or not; the same gate as the shard prover.
pub fn aggregation_card(cards: &[CardState]) -> Option<&CardState> {
cards.iter().filter(|c| c.vendor == "nvidia" && c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).max_by_key(|c| c.vram_mb)
}
/// `os` is `std::env::consts::OS` ("windows", "linux", "macos"); `wsl_answers` is read on Windows only; `ram_mb` is the
/// machine's RAM when the platform reports it (None = unknown, no gate).
pub fn decide(cards: &[CardState], os: &str, wsl_answers: Option<bool>, ram_mb: Option<u64>) -> Decision {
let nvidia: Vec<&CardState> = cards.iter().filter(|c| c.vendor == "nvidia").collect();
// a mining card needs 20 GB (the measured mine-and-prove peak of 16.8 GB), a card that only proves 16 GB
let able: Vec<&CardState> = nvidia.iter().copied().filter(|c| c.vram_mb >= if c.enabled { MIN_VRAM_MB_MINING } else { MIN_VRAM_MB_PROVE_ONLY }).collect();
let off = |line: String| Decision { on: false, line };
if os == "macos" {
return off("proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)".into());
}
let Some(best) = able.iter().max_by_key(|c| c.vram_mb) else {
let seen = if nvidia.is_empty() {
"no NVIDIA card".to_string()
} else {
nvidia.iter().map(|c| format!("{} {} GB{}", c.name, gb(c.vram_mb), if c.enabled { ", mining" } else { "" })).collect::<Vec<_>>().join(", ")
};
let why = if nvidia.iter().any(|c| c.vram_mb >= 15_872) {
"a full shard needs a 24 GB card (measured 20.4 GB on the adopted shard size, 13.9 GB for an empty one); this card is under that, so Settings would switch proving on at your own risk"
} else if nvidia.is_empty() {
"this machine mines and does not prove: no zkVM proves on an AMD GPU today, and the CPU prover costs about 5 minutes a shard at a 30 GB RSS (bench-log, the SP1 CPU prover on PC 1); proving needs an NVIDIA card with 24 GB or more"
} else {
"no NVIDIA card with 24 GB or more (the GPU prover's floor is 13.9 GB for an empty shard and 20.4 GB for a full one)"
};
return off(format!("proving off by default: {why} ({seen})"));
};
let card = format!("{} ({} GB{})", best.name, gb(best.vram_mb), if best.enabled { ", mining too" } else { ", proving only" });
if os == "windows" {
if let Some(ram) = ram_mb {
if ram < MIN_RAM_MB_WINDOWS {
return off(format!("proving off by default: {card} qualifies but this PC has {} GB of RAM; proving needs 32 GB on Windows until a smaller PC is measured (the WSL2 prover held 7.9 GB on a 63 GB PC); Settings switches it on", gb(ram)));
}
}
}
let size_note = if best.vram_mb >= VRAM_MB_PROTOTYPE_SHARD { "" } else { "; until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" };
match os {
"windows" => match wsl_answers {
Some(true) => Decision { on: true, line: format!("proving on by default: {card} with WSL2 (Ubuntu-24.04 answers){size_note}; Settings switches it off") },
_ => off(format!("proving off: {card} qualifies but WSL2 (Ubuntu-24.04) did not answer; Set up installs it, then Settings switches proving on")),
},
"linux" => Decision { on: true, line: format!("proving on by default: {card} on Linux (the host runs next to the engine){size_note}; Settings switches it off") },
other => off(format!("proving off: {card} on {other}, no prover path there; Settings switches it on")),
}
}
#[cfg(test)]
mod tests {
use super::*;
fn card(vendor: &str, name: &str, vram_mb: u64) -> CardState {
CardState { vendor: vendor.into(), name: name.into(), vram_mb, enabled: true, ..Default::default() }
}
fn idle(vendor: &str, name: &str, vram_mb: u64) -> CardState {
CardState { enabled: false, ..card(vendor, name, vram_mb) }
}
#[test]
fn a_5090_with_wsl2_on_windows_is_on() {
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607), card("amd", "AMD Radeon(TM) Graphics", 512)], "windows", Some(true), Some(63_132));
assert!(d.on);
assert!(d.line.starts_with("proving on by default: NVIDIA GeForce RTX 5090 (32 GB, mining too) with WSL2"), "{}", d.line);
}
#[test]
fn windows_without_wsl2_is_off_with_the_setup_hint() {
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(false), Some(65_000));
assert!(!d.on);
assert!(d.line.contains("did not answer") && d.line.contains("Set up"), "{}", d.line);
assert!(!decide(&[card("nvidia", "RTX 4090", 24_564)], "windows", None, Some(65_000)).on, "an unread probe is not an answer");
}
#[test]
fn linux_needs_no_wsl2_and_the_memory_gates_hold() {
// the tiers of the S_p curve: 32 GB on with no note; 24 GB on with the prototype-size note; 16 GB and 12 GB off
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5090", 32_607)], "linux", None, None);
assert!(d.on && !d.line.contains("fee switch"), "{}", d.line);
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None);
assert!(d.on && d.line.contains("needs 32 GB, so this card proves from the switch on"), "{}", d.line);
assert!(decide(&[idle("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, None).on, "mining or not, 24 GB proves");
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None);
assert!(!d.on);
assert!(d.line.contains("a full shard needs a 24 GB card") && d.line.contains("RTX 5080 16 GB, mining"), "{}", d.line);
assert!(!decide(&[idle("nvidia", "NVIDIA GeForce RTX 5080", 16_303)], "linux", None, None).on, "16 GB holds no full shard even alone");
let d = decide(&[idle("nvidia", "NVIDIA GeForce RTX 3060", 12_288)], "linux", None, None);
assert!(!d.on);
assert!(d.line.contains("no NVIDIA card with 24 GB or more") && d.line.contains("RTX 3060 12 GB"), "{}", d.line);
assert!(!decide(&[card("nvidia", "NVIDIA GeForce RTX 3080", 10_240)], "linux", None, None).on);
let d = decide(&[card("amd", "Radeon RX 9070 XT", 16_384)], "linux", None, None);
assert!(!d.on && d.line.contains("mines and does not prove"), "{}", d.line);
assert!(decide(&[], "linux", None, None).line.contains("mines and does not prove"));
}
#[test]
fn apple_silicon_stays_off() {
let d = decide(&[card("apple", "Apple M5 Max", 65_536)], "macos", None, Some(65_536));
assert!(!d.on);
assert!(d.line.contains("Apple silicon"));
assert!(!decide(&[card("nvidia", "RTX 5090", 32_607)], "macos", Some(true), None).on, "the OS rule comes first");
}
#[test]
fn a_windows_pc_under_32_gb_stays_off_and_the_aggregation_card_follows_the_same_gate() {
let d = decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), Some(16_300));
assert!(!d.on);
assert!(d.line.contains("16 GB of RAM") && d.line.contains("needs 32 GB on Windows"), "{}", d.line);
assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "windows", Some(true), None).on, "unknown RAM is not a gate");
assert!(decide(&[card("nvidia", "NVIDIA GeForce RTX 4090", 24_564)], "linux", None, Some(16_300)).on, "the RAM gate is Windows only (the WSL2 VM)");
let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4070 Ti", 12_282)];
assert!(aggregation_card(&cards).is_none(), "a mining 16 GB card and an idle 12 GB card cannot aggregate");
let cards = [card("nvidia", "RTX 5080", 16_303), idle("nvidia", "RTX 4090", 24_564)];
assert_eq!(aggregation_card(&cards).map(|c| c.name.as_str()), Some("RTX 4090"));
let cards = [card("nvidia", "RTX 5090", 32_607)];
assert_eq!(aggregation_card(&cards).map(|c| c.vram_mb), Some(32_607));
}
#[test]
fn the_biggest_qualifying_card_is_named() {
let d = decide(&[idle("nvidia", "RTX 4090", 24_564), card("nvidia", "RTX 5090", 32_607)], "linux", None, None);
assert!(d.line.contains("RTX 5090 (32 GB, mining too)"), "{}", d.line);
}
}

View file

@ -47,6 +47,8 @@ pub struct Work {
pub shard_wei: u128,
/// The first of this machine's keys that is assigned (the label that signs).
pub key_hash: String,
/// The chain block's DAA score (the deadline clock of spec 7.8).
pub daa: u64,
}
/// Parses the node's work list. Newest first, as the node returns it.
@ -67,6 +69,7 @@ pub fn parse_work(v: &Value) -> Vec<Work> {
in_pool: w["pool"].as_array().map(|p| !p.is_empty()).unwrap_or(false),
shard_wei: hexu(&w["shardWei"]),
key_hash: w["assignedKeys"].as_array().and_then(|k| k.first()).and_then(|k| k.as_str()).unwrap_or("").to_string(),
daa: hexu(&w["daaScore"]) as u64,
})
.collect()
})
@ -288,6 +291,44 @@ fn set<F: FnOnce(&mut crate::state::ProvingState)>(shared: &Shared, f: F) {
f(&mut st.proving);
}
/// The pinned ids out of `igneum-prove-host --mode id` ("RESULT id: pinned guests: shard program id 0x... (...)
/// aggregator id 0x... (...)"): (shard program id, aggregator id). None when the line is not there.
pub fn ids_from_describe(text: &str) -> Option<(String, String)> {
let line = text.lines().find(|l| l.contains("shard program id "))?;
let after = |key: &str| -> Option<String> {
let rest = line.split(key).nth(1)?.trim_start();
let id: String = rest.chars().take_while(|c| c.is_ascii_alphanumeric()).collect();
(id.starts_with("0x") && id.len() == 66).then_some(id)
};
Some((after("shard program id ")?, after("aggregator id ")?))
}
/// Reads the pinned ids once (`--mode id` does no key setup, under a second) and puts them on the state. The Prove
/// page shows them with the verifier state, proving on or off.
fn read_ids(shared: &Shared, t: &Tools) {
if !shared.state.lock().unwrap().proving.program_id.is_empty() {
return;
}
// not run_tool: that stops the child while proving is off, and the ids are wanted proving on or off
let out = if t.wsl {
let body = format!("export RUST_LOG=off\nexec {} --mode id", crate::wslhost::sq(&t.host.display().to_string()));
let Ok(file) = crate::wslhost::write_script("prove-ids", &body) else { return };
crate::detect::run_timeout(&mut crate::wslhost::command(&crate::platform::tool("wsl"), crate::wslhost::DISTRO, None, &file.path, true, &[]), None, Duration::from_secs(60))
} else {
crate::detect::run_timeout(Command::new(&t.host).args(["--mode", "id"]).env("RUST_LOG", "off"), None, Duration::from_secs(60))
};
match ids_from_describe(&out.unwrap_or_default()) {
Some((shard, agg)) => {
shared.log(&format!("prover: pinned shard program id {shard}, aggregator id {agg}"));
set(shared, |p| {
p.program_id = shard;
p.aggregator_id = agg;
});
}
None => shared.log("prover: --mode id printed no pinned ids (an older host)"),
}
}
static BIN_DIR: std::sync::OnceLock<PathBuf> = std::sync::OnceLock::new();
/// Starts the prover thread. It idles while the setting is off or the node is not synced.
@ -301,11 +342,27 @@ pub fn start(shared: Arc<Shared>, bin_dir: PathBuf) {
fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
let mut attempted: HashSet<(String, u32)> = HashSet::new();
let mut attempted_segments: HashSet<u64> = HashSet::new();
let mut tools: Option<Tools> = None;
let ram_mb = crate::detect::total_ram_mb();
let mut last_probe = Instant::now() - Duration::from_secs(600);
let mut submitted: Vec<(u64, String, u32, u128)> = Vec::new();
// proving v1 segment path: (first, last, aggregator wei) of the segment records this machine submitted
let mut submitted_segments: Vec<(u64, u64, u128)> = Vec::new();
let mut last_segment_secs: Option<f64> = None;
// segment records the node refused by the chain rule ("does not chain to ... pending"): held and offered again
// every pass until the segment's deadline (the fresh-record window of spec 7.8 is the segment length in DAA on
// the rule as shipped, 6 October 2026; from the fresh-rule switch the first retry lands)
let mut held_segments: Vec<HeldSegment> = Vec::new();
let mut last_verifier_read = Instant::now() - Duration::from_secs(600);
let mut asked_restart = false;
// macOS and Linux: the host sits next to the engine, so its pinned ids are read at once, proving on or off
// (Windows runs the host inside WSL2, which is probed only once proving is on)
if !cfg!(windows) {
if let Ok(t) = find_tools(&bin_dir) {
read_ids(&shared, &t);
}
}
loop {
std::thread::sleep(Duration::from_secs(10));
let enabled = shared.settings.lock().unwrap().prove;
@ -345,6 +402,7 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
asked_restart = true;
shared.send(crate::engine::Cmd::RestartNode("the WSL2 prover is installed now; the node restarts to verify proof records".into()));
}
read_ids(&shared, &t);
tools = Some(t);
}
Err(e) => {
@ -360,6 +418,20 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
}
}
let Some(t) = tools.as_ref() else { continue };
// consequences review C22 (5 October 2026): the SP1 CPU prover takes 29.5 to 30.5 GB of RSS and about five
// minutes a shard whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1"); on a machine under
// 32 GB it would swap the node out, so the CPU path is refused here, Settings or not, with the reason
if !t.cuda {
if let Some(ram) = ram_mb {
if ram < crate::provedefault::MIN_RAM_MB_WINDOWS {
set(&shared, |p| {
p.status = "off".into();
p.message = format!("the CPU prover needs 32 GB of RAM (30 GB measured on PC 1); this machine has {} GB, so proving stays off here", (ram + 512) / 1024);
});
continue;
}
}
}
set(&shared, |p| {
p.enabled = true;
p.available = true;
@ -388,7 +460,7 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
});
continue;
}
let work = match evm_rpc(&shared, "igneum_getAssignedShards", json!([keys.iter().map(|(_, h)| h.clone()).collect::<Vec<_>>(), 60]), Duration::from_secs(10)) {
let work = match evm_rpc(&shared, "igneum_getAssignedShards", json!([keys.iter().map(|(_, h)| h.clone()).collect::<Vec<_>>(), crate::segments::WORK_LOOKBACK]), Duration::from_secs(10)) {
Ok(v) => parse_work(&v),
Err(e) => {
set(&shared, |p| {
@ -415,15 +487,125 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
});
}
}
// held segment records: offered again, dropped past the deadline
if !held_segments.is_empty() {
let tip_daa = evm_rpc(&shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10)).ok().and_then(|st| st["tipDaa"].as_str().and_then(|x| u64::from_str_radix(x.trim_start_matches("0x"), 16).ok())).unwrap_or(0);
let mut keep = Vec::new();
for h in held_segments.drain(..) {
match retry_held(&shared, &h, tip_daa) {
Retry::Accepted => {
shared.event("proving", &format!("segment {}..{} record accepted on retry {} (held {} s)", h.first, h.last, h.tries + 1, h.since.elapsed().as_secs()));
submitted_segments.push((h.first, h.last, h.agg_wei));
set(&shared, |p| {
p.segments_submitted += 1;
p.aggregated += 1;
p.segment_note = format!("segment {}..{} accepted on retry", h.first, h.last);
});
}
Retry::Expired(why) => {
shared.log(&format!("prover: segment {}..{} record dropped after {} tries: {why}", h.first, h.last, h.tries));
}
Retry::Again(why) => {
let mut h = h;
h.tries += 1;
if h.tries % 30 == 1 {
shared.log(&format!("prover: segment {}..{} record held (try {}): {why}", h.first, h.last, h.tries));
}
keep.push(h);
}
}
}
held_segments = keep;
set(&shared, |p| p.segments_held = held_segments.len() as u32);
}
// paid segments among what we submitted
for (first, last, wei) in submitted_segments.clone() {
let paid = evm_rpc(&shared, "igneum_getSegmentRecords", json!([format!("{first:#x}")]), Duration::from_secs(10))
.ok()
.map(|r| !r["paid"].is_null() && r["paid"]["payout"].as_str().map(|a| a.eq_ignore_ascii_case(&payout_address(&shared))).unwrap_or(false))
.unwrap_or(false);
if paid {
submitted_segments.retain(|x| x.0 != first);
shared.event("proving", &format!("segment {first}..{last} paid {} IGN to the aggregator", wei as f64 / 1e18));
set(&shared, |p| {
p.segments_paid += 1;
p.segment_paid_wei += wei;
});
}
}
let assigned = work.iter().filter(|w| w.assigned).count() as u32;
set(&shared, |p| {
p.assigned = assigned;
p.keys = keys.len() as u32;
});
// proving v1 (spec 7.8): the aggregator step, when the node says v1 is active; one attempt a pass
if let Some((label0, _)) = keys.first() {
match aggregate_once(&shared, t, label0, &payout_address(&shared), &mut attempted_segments) {
Ok(Some(msg)) => {
shared.log(&format!("aggregator: {msg}"));
set(&shared, |p| p.segment_note = msg);
}
Ok(None) => {}
Err(e) => {
shared.log(&format!("aggregator: {e}"));
set(&shared, |p| p.segment_note = e);
}
}
}
// proving v1 segment path (src/segments.rs, 6 October 2026): a whole segment first, the newest shard only
// when no whole segment qualifies
let payout = shared.settings.lock().unwrap().address.clone();
if payout.len() == 42 {
if let Some((seg, prev_file, expected_pv)) = pick_segment(&shared, &work, &keys[0].1, &mut attempted_segments, last_segment_secs) {
attempted_segments.insert(seg.first);
let started = Instant::now();
match prove_segment(&shared, t, &seg, &keys[0].0, &payout, prev_file.as_deref(), &expected_pv, &mut submitted) {
Ok(SegmentOutcome::Held(h)) => {
let secs = started.elapsed().as_secs_f64();
last_segment_secs = Some(secs);
shared.event("proving", &format!("segment {}..{}: {} shards proven and submitted in {secs:.0} s; the segment record is held ({})", seg.first, seg.last, seg.shards.len(), h.why));
set(&shared, |p| {
p.segment_last_s = secs;
p.status = "submitted".into();
p.message = format!("segment {}..{}: shards submitted, the segment record waits for the chain rule", seg.first, seg.last);
p.segment_note = format!("segment {}..{} proven whole in {secs:.0} s; its record is held: {}", seg.first, seg.last, h.why);
p.current = String::new();
});
held_segments.push(h);
set(&shared, |p| p.segments_held = held_segments.len() as u32);
}
Ok(SegmentOutcome::Submitted(agg_wei)) => {
let secs = started.elapsed().as_secs_f64();
last_segment_secs = Some(secs);
submitted_segments.push((seg.first, seg.last, agg_wei));
shared.event("proving", &format!("segment {}..{}: {} shards proven, aggregated and submitted in {secs:.0} s", seg.first, seg.last, seg.shards.len()));
set(&shared, |p| {
p.segments_submitted += 1;
p.segment_last_s = secs;
p.status = "submitted".into();
p.message = format!("segment {}..{} submitted; paid when a block carries it", seg.first, seg.last);
p.segment_note = format!("segment {}..{} proven whole in {secs:.0} s", seg.first, seg.last);
p.current = String::new();
});
}
Err(e) => {
shared.log(&format!("prover: segment {}..{}: {e}", seg.first, seg.last));
set(&shared, |p| {
p.failed += 1;
p.status = "idle".into();
p.message = e.clone();
p.segment_note = e;
p.current = String::new();
});
}
}
continue;
}
}
let Some(w) = choose(&work, &attempted) else {
set(&shared, |p| {
p.status = if submitted.is_empty() { "idle".into() } else { "submitted".into() };
p.message = if assigned == 0 { "no shard assigned to this machine and none open in the last 60 blocks".into() } else { "every assigned and open shard is proven or paid".into() };
p.message = if assigned == 0 { format!("no shard assigned to this machine and none open in the last {} blocks", crate::segments::WORK_LOOKBACK) } else { "every assigned and open shard is proven or paid".into() };
});
continue;
};
@ -459,11 +641,15 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
if !ok || !fixture.exists() {
return Err(format!("exporter: {}", out.lines().rev().find(|l| !l.trim().is_empty()).unwrap_or("failed")));
}
set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "proving on the CPU (slow)".into() });
set(&shared, |p| p.message = if t.cuda { "proving on the GPU".into() } else { "CPU prover: about five minutes a shard, 30 GB of RAM, paid only when no card proves first".into() });
let prover_env = if t.cuda { "cuda" } else { "cpu" };
let (ok, out) = run_tool(&shared, t, &t.host, &[fix_p, "--mode".into(), "compressed".into(), "--shard".into(), w.shard.to_string(), "--prover".into(), payout.clone(), "--out".into(), res_p], &[("SP1_PROVER", prover_env), ("RUST_LOG", "off")], Duration::from_secs(3 * 3600), &dir.join(format!("prove-{}-{}.log", w.number, w.shard)));
if !ok || !results.exists() {
return Err(format!("prover: {}", out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed")));
let last = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string();
// the root-socket class (5 October 2026, PC 2 at 20:00Z and 21:25Z): a job that ran the host as root
// inside WSL2 left /tmp/sp1-cuda-0.sock owned by root, and this user's client cannot open it
let hint = if last.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user, left by a job that ran the prover as root: remove it as that user, or run the socket-fix job)" } else { "" };
return Err(format!("prover: {last}{hint}"));
}
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
let statement = res["statement"].as_str().ok_or("no statement in the results")?.to_string();
@ -512,6 +698,336 @@ fn loop_forever(shared: Arc<Shared>, bin_dir: PathBuf) {
}
}
/// A segment record the node refused by the chain rule, kept with its proof for another offer.
pub struct HeldSegment {
pub first: u64,
pub last: u64,
pub deadline_daa: u64,
pub record: String,
pub proof_path: PathBuf,
pub agg_wei: u128,
pub why: String,
pub tries: u32,
pub since: Instant,
}
pub enum SegmentOutcome {
Submitted(u128),
Held(HeldSegment),
}
pub enum Retry {
Accepted,
Again(String),
Expired(String),
}
/// Offers a held segment record again: accepted, held for another pass, or dropped past the segment's deadline.
fn retry_held(shared: &Shared, h: &HeldSegment, tip_daa: u64) -> Retry {
if tip_daa > 0 && tip_daa + 1 > h.deadline_daa {
return Retry::Expired(format!("past the deadline DAA {} at tip DAA {tip_daa}", h.deadline_daa));
}
let Ok(proof) = std::fs::read(&h.proof_path) else { return Retry::Expired(format!("proof file {} gone", h.proof_path.display())) };
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
match evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": h.record, "proof": proof_hex }]), Duration::from_secs(60)) {
Ok(r) if r["accepted"].as_bool().unwrap_or(false) => Retry::Accepted,
Ok(r) => Retry::Again(r["reason"].as_str().unwrap_or("?").to_string()),
Err(e) => Retry::Again(e),
}
}
/// Proving v1 segment path, the choice: the node's v1 status (active, the grid start, the segment length, the
/// deadline clock), the work list grouped into whole untouched segments (`segments::whole_segments`), the
/// candidates inside the deadline ranked for this key, then for the best three the node's segment statement:
/// executed and pending; the previous segment either paid with its proof in this node's pool (the chain continues,
/// `--prev`) or not paid and with no verified record of it waiting in the pool (fresh). Returns the segment, the
/// previous proof's host path when the chain continues, and the public values the node expects.
fn pick_segment(shared: &Shared, work: &[Work], key_hash: &str, attempted: &mut HashSet<u64>, last_secs: Option<f64>) -> Option<(crate::segments::SegmentWork, Option<String>, String)> {
let hexu = |x: &Value| x.as_str().and_then(|s| u64::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0);
let st = evm_rpc(shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10)).ok()?;
let v1 = &st["v1"];
if !v1["active"].as_bool().unwrap_or(false) || v1["start"].is_null() {
return None;
}
let (start, n, unproven, tip_daa) = (hexu(&v1["start"]), hexu(&v1["segmentBlocks"]).max(1), hexu(&v1["unprovenDaa"]), hexu(&st["tipDaa"]));
let segs = crate::segments::whole_segments(start, n, unproven, work);
let need = crate::segments::need_daa(last_secs);
let cands = crate::segments::candidates(&segs, tip_daa, need, key_hash, attempted);
if cands.is_empty() {
return None;
}
let dir = shared.runtime.app_dir.join("proving");
let _ = std::fs::create_dir_all(&dir);
let wsl = cfg!(windows);
let as_host_path = |p: &Path| if wsl { wsl_path(p) } else { p.display().to_string() };
for seg in cands.into_iter().take(3) {
let Ok(stmt) = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{:#x}", seg.first)]), Duration::from_secs(10)) else { continue };
if !stmt["executed"].as_bool().unwrap_or(false) || stmt["status"]["status"].as_str() != Some("pending") {
attempted.insert(seg.first);
continue;
}
let prev = &stmt["previous"];
if prev.is_null() {
// fresh only when no record of the previous segment is waiting to be carried (the chain rule would
// refuse a fresh record once that one pays)
if seg.first >= start + n {
let p = evm_rpc(shared, "igneum_getSegmentRecords", json!([format!("{:#x}", seg.first - n)]), Duration::from_secs(10)).unwrap_or(Value::Null);
let waiting = p["pool"].as_array().map(|a| a.iter().any(|e| e["verified"] == json!(true) && e["includedIn"].is_null())).unwrap_or(false);
if waiting || !p["paid"].is_null() {
continue;
}
}
return Some((seg, None, stmt["publicValuesFresh"].as_str().unwrap_or("").to_string()));
}
if prev["proofInPool"] != json!(true) {
continue;
}
let Ok(got) = evm_rpc(shared, "igneum_getSegmentProofBytes", json!([prev["first"], prev["keyHash"]]), Duration::from_secs(60)) else { continue };
let hex = got["proof"].as_str().unwrap_or("").trim_start_matches("0x").to_string();
if hex.is_empty() {
continue;
}
let bytes: Vec<u8> = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect();
let f = dir.join(format!("prev-{}.bin", seg.first));
if std::fs::write(&f, bytes).is_err() {
continue;
}
return Some((seg, Some(as_host_path(&f)), stmt["publicValuesContinuing"].as_str().unwrap_or("").to_string()));
}
None
}
/// Proving v1 segment path, the work: one export of the chain to the segment's last block, one fixture per block,
/// one host run (`--mode chain --save-shards`, `--prev` when the chain continues) that proves every shard and
/// aggregates the segment, then every shard record signed and submitted (the shard payouts) and the segment
/// record signed and submitted (the aggregator share). Returns the segment's aggregator wei.
fn prove_segment(shared: &Shared, t: &Tools, seg: &crate::segments::SegmentWork, label: &str, payout: &str, prev_file: Option<&str>, expected_pv: &str, submitted: &mut Vec<(u64, String, u32, u128)>) -> Result<SegmentOutcome, String> {
let (first, last) = (seg.first, seg.last);
let dir = shared.runtime.app_dir.join("proving").join(format!("seg-{first}"));
let _ = std::fs::create_dir_all(&dir);
let as_host_path = |p: &Path| if t.wsl { wsl_path(p) } else { p.display().to_string() };
set(shared, |p| {
p.status = "proving".into();
p.current = format!("segment {first}..{last} ({} shards)", seg.shards.len());
p.started_at = crate::platform::unix_now_f();
p.message = "exporting the chain and cutting the segment's blocks".into();
});
shared.log(&format!("prover: segment {first}..{last} claimed ({} shards{}): export, cut, chain ({}), sign, submit", seg.shards.len(), if prev_file.is_some() { ", continuing the previous segment's proof" } else { ", fresh" }, if t.cuda { "CUDA" } else { "CPU" }));
// 1. export once, cut every block
let seq = dir.join("seq.json");
let export = evm_rpc(shared, "igneum_exportSegments", json!(["0x0", format!("{last:#x}")]), Duration::from_secs(300))?;
std::fs::write(&seq, export.to_string()).map_err(|e| e.to_string())?;
let mut fixtures: Vec<String> = Vec::new();
for b in first..=last {
let fixture = dir.join(format!("block-{b}.json"));
let (ok, out) = run_tool(shared, t, &t.export, &[as_host_path(&seq), b.to_string(), as_host_path(&fixture)], &[], Duration::from_secs(600), &dir.join(format!("export-{b}.log")));
if !ok || !fixture.exists() {
return Err(format!("exporter, block {b}: {}", out.lines().rev().find(|l| !l.trim().is_empty()).unwrap_or("failed")));
}
fixtures.push(as_host_path(&fixture));
}
let _ = std::fs::remove_file(&seq);
// 2. the chain: every shard proven, every block aggregated with the previous, in one process
set(shared, |p| p.message = format!("proving {} shards and aggregating segment {first}..{last} ({})", seg.shards.len(), if t.cuda { "GPU" } else { "CPU, slow" }));
let results = dir.join("chain-results.json");
let mut args: Vec<String> = vec!["--mode".into(), "chain".into(), "--chain".into(), fixtures.join(","), "--prover".into(), payout.to_string(), "--save-shards".into(), "--out".into(), as_host_path(&results)];
if let Some(pf) = prev_file {
args.push("--prev".into());
args.push(pf.to_string());
}
let (ok, out) = run_tool(shared, t, &t.host, &args, &[("SP1_PROVER", if t.cuda { "cuda" } else { "cpu" }), ("RUST_LOG", "off")], Duration::from_secs(3 * 3600), &dir.join("chain.log"));
if !ok || !results.exists() {
let last_line = out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed").to_string();
let hint = if last_line.contains("PermissionDenied") { " (a GPU-server socket /tmp/sp1-cuda-*.sock owned by another user: the root-socket class)" } else { "" };
return Err(format!("chain: {last_line}{hint}"));
}
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
let to_win = |f: &str| if t.wsl { PathBuf::from(f.replace("/mnt/c/", "C:/")) } else { PathBuf::from(f) };
// 3. the shard records
let chain = chain_name(shared);
let mut shard_ok = 0usize;
for b in res["blocks"].as_array().cloned().unwrap_or_default() {
for r in b["shard_records"].as_array().cloned().unwrap_or_default() {
let (number, hash, shard) = (r["number"].as_u64().unwrap_or(0), r["block_hash"].as_str().unwrap_or("").to_string(), r["shard"].as_u64().unwrap_or(0) as u32);
let statement = r["statement"].as_str().unwrap_or("").to_string();
let proof_sha = r["proof_sha256"].as_str().unwrap_or("").to_string();
let proof_file = r["proof_file"].as_str().unwrap_or("").to_string();
let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-record", label, &chain, &hash, &number.to_string(), &shard.to_string(), payout, &statement, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-record did not run")?;
let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-record: {}", sg.trim()))?;
let record = signed["record"].as_str().ok_or("sign-record gave no record")?.to_string();
let proof = std::fs::read(to_win(&proof_file)).map_err(|e| format!("proof file {proof_file}: {e}"))?;
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
let out = evm_rpc(shared, "igneum_submitProofRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?;
if out["accepted"].as_bool().unwrap_or(false) {
shard_ok += 1;
let wei = seg.shards.iter().position(|(n, _, s)| *n == number && *s == shard).map(|_| seg.shard_wei / seg.shards.len().max(1) as u128).unwrap_or(0);
submitted.push((number, hash.clone(), shard, wei));
} else {
shared.log(&format!("prover: segment {first}..{last}: block {number} shard {shard} record refused: {}", out["reason"].as_str().unwrap_or("?")));
}
}
}
if shard_ok != seg.shards.len() {
return Err(format!("{shard_ok} of {} shard records accepted; the segment record is not submitted", seg.shards.len()));
}
set(shared, |p| {
p.proved += shard_ok as u32;
p.submitted += shard_ok as u32;
});
// 4. the segment record: the aggregated statement against the node's native one (every field but provers)
let pv = res["segment_public_values"].as_str().ok_or("no public values in the chain results")?.to_string();
let proof_sha = res["segment_proof_sha256"].as_str().ok_or("no segment proof hash in the chain results")?.to_string();
let proof_file = res["segment_proof_file"].as_str().ok_or("no segment proof file in the chain results")?.to_string();
let strip = |h: &str| { let h = h.trim_start_matches("0x"); if h.len() == 680 { format!("{}{}", &h[..472], &h[536..]) } else { h.to_string() } };
if strip(&pv) != strip(expected_pv) {
return Err(format!("the aggregated statement differs from the node's native statement (it would be vetoed); ours {} node {}", &pv[..66.min(pv.len())], &expected_pv[..66.min(expected_pv.len())]));
}
let last_hash = seg.shards.iter().rev().find(|(n, _, _)| *n == last).map(|(_, h, _)| h.clone()).ok_or("no last block hash")?;
let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-segment-record", label, &chain, &first.to_string(), &last.to_string(), &last_hash, payout, &pv, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-segment-record did not run")?;
let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-segment-record: {}", sg.trim()))?;
let record = signed["record"].as_str().ok_or("sign-segment-record gave no record")?.to_string();
let proof = std::fs::read(to_win(&proof_file)).map_err(|e| format!("segment proof file {proof_file}: {e}"))?;
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
let r = evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?;
let hexu = |x: &Value| x.as_str().and_then(|s| u128::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0);
let stmt = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{first:#x}")]), Duration::from_secs(10)).unwrap_or(Value::Null);
let agg_wei = hexu(&stmt["aggregatorWei"]);
// the fixtures and the export go; the proofs stay (the next segment's chain link, and a held record's offer)
for b in first..=last {
let _ = std::fs::remove_file(dir.join(format!("block-{b}.json")));
}
if !r["accepted"].as_bool().unwrap_or(false) {
let why = r["reason"].as_str().unwrap_or("?").to_string();
// the chain rule's refusal ("does not chain to ... pending until DAA ..."): held, not failed; anything else
// (a bad statement, a late carrier) is an error
if why.contains("does not chain") {
let deadline = hexu(&stmt["status"]["deadline_daa"]) as u64;
return Ok(SegmentOutcome::Held(HeldSegment { first, last, deadline_daa: if deadline > 0 { deadline } else { u64::MAX }, record, proof_path: to_win(&proof_file), agg_wei, why, tries: 0, since: Instant::now() }));
}
return Err(format!("segment record refused: {why}"));
}
set(shared, |p| p.aggregated += 1);
Ok(SegmentOutcome::Submitted(agg_wei))
}
/// Proving v1 (spec 7.8): one aggregation attempt. When the node reports v1 active, takes the newest executed
/// segment that is still pending and not yet attempted here, needs one shard proof per shard of every block in
/// this node's pool (`igneum_getProofBytes`, a verified one when there is one) and, when the previous segment is
/// proven, its aggregated proof (`igneum_getSegmentProofBytes`); runs `igneum-prove-host --mode aggregate` over the
/// run of blocks (one process, one key setup), checks the public values against the node's native statement
/// (every field but `provers`), signs the record with the first key's label and submits it. Returns a line for
/// the log and the tile, or None when there is nothing to do.
fn aggregate_once(shared: &Shared, t: &Tools, label: &str, payout: &str, attempted: &mut HashSet<u64>) -> Result<Option<String>, String> {
let hexu = |x: &Value| x.as_str().and_then(|s| u64::from_str_radix(s.trim_start_matches("0x"), 16).ok()).unwrap_or(0);
let st = evm_rpc(shared, "igneum_getProvingStatus", json!([]), Duration::from_secs(10))?;
let v1 = &st["v1"];
if !v1["active"].as_bool().unwrap_or(false) || v1["start"].is_null() {
return Ok(None);
}
// the card gate (consequences review C2): a chained aggregation peaked at 16,751 MiB with the miner resident; the
// prover's own 24 GB gate applies; without such a card this machine proves shards and never aggregates
{
let cards = shared.state.lock().unwrap().mining.cards.clone();
if crate::provedefault::aggregation_card(&cards).is_none() {
return Ok(Some("no aggregation on this machine: it needs a 24 GB card (16.8 GB measured with the miner resident); shards still prove".into()));
}
}
let tip = hexu(&evm_rpc(shared, "eth_blockNumber", json!([]), Duration::from_secs(10))?);
let mut seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{tip:#x}")]), Duration::from_secs(10))?;
if !seg["executed"].as_bool().unwrap_or(false) {
let first = hexu(&seg["first"]);
if first == 0 || first - 1 < hexu(&v1["start"]) {
return Ok(None);
}
seg = evm_rpc(shared, "igneum_getSegmentStatement", json!([format!("{:#x}", first - 1)]), Duration::from_secs(10))?;
}
let (first, last) = (hexu(&seg["first"]), hexu(&seg["last"]));
if seg["status"]["status"].as_str() != Some("pending") || attempted.contains(&first) || payout.len() != 42 {
return Ok(None);
}
let dir = shared.runtime.app_dir.join("proving").join(format!("seg-{first}"));
let _ = std::fs::create_dir_all(&dir);
let as_host_path = |p: &Path| if t.wsl { wsl_path(p) } else { p.display().to_string() };
// the shard proofs, one per shard of every block, from this node's pool
let mut groups: Vec<String> = Vec::new();
let mut missing: Vec<String> = Vec::new();
for b in seg["blocks"].as_array().cloned().unwrap_or_default() {
let n = hexu(&b["number"]);
let shards = b["shards"].as_u64().unwrap_or(0) as u32;
let have = b["shardProofs"].as_array().cloned().unwrap_or_default();
let mut files = Vec::new();
for i in 0..shards {
let pick = have.iter().find(|e| e["shard"].as_u64() == Some(i as u64) && e["verified"] == json!(true)).or_else(|| have.iter().find(|e| e["shard"].as_u64() == Some(i as u64)));
let Some(e) = pick else {
missing.push(format!("{n}/{i}"));
continue;
};
let got = evm_rpc(shared, "igneum_getProofBytes", json!([format!("{n:#x}"), i, e["keyHash"]]), Duration::from_secs(60))?;
let hex = got["proof"].as_str().ok_or("no proof bytes")?.trim_start_matches("0x").to_string();
let bytes: Vec<u8> = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect();
let f = dir.join(format!("b{n}-s{i}.bin"));
std::fs::write(&f, bytes).map_err(|e| e.to_string())?;
files.push(as_host_path(&f));
}
groups.push(files.join(","));
}
if !missing.is_empty() {
return Ok(Some(format!("segment {first}..{last}: waiting for shard proofs {} in this node's pool", missing.join(" "))));
}
// the previous segment's aggregated proof, when the chain continues
let prev = &seg["previous"];
let (prev_file, expected_pv) = if prev.is_null() {
(None, seg["publicValuesFresh"].as_str().unwrap_or("").to_string())
} else if prev["proofInPool"] == json!(true) {
let got = evm_rpc(shared, "igneum_getSegmentProofBytes", json!([prev["first"], prev["keyHash"]]), Duration::from_secs(60))?;
let hex = got["proof"].as_str().ok_or("no segment proof bytes")?.trim_start_matches("0x").to_string();
let bytes: Vec<u8> = (0..hex.len() / 2).map(|k| u8::from_str_radix(&hex[2 * k..2 * k + 2], 16).unwrap_or(0)).collect();
let f = dir.join("prev-aggregated.bin");
std::fs::write(&f, bytes).map_err(|e| e.to_string())?;
(Some(as_host_path(&f)), seg["publicValuesContinuing"].as_str().unwrap_or("").to_string())
} else {
return Ok(Some(format!("segment {first}..{last}: the previous segment's proof is not in this node's pool; waiting")));
};
attempted.insert(first);
let parent = seg["blocks"][0]["parentHash"].as_str().ok_or("no parent hash")?.to_string();
let last_hash = seg["blocks"].as_array().and_then(|a| a.last()).and_then(|b| b["hash"].as_str()).ok_or("no last hash")?.to_string();
let results = dir.join("results.json");
let started = Instant::now();
set(shared, |p| p.message = format!("aggregating segment {first}..{last} ({})", if t.cuda { "GPU" } else { "CPU, slow" }));
let mut args: Vec<String> = vec!["--mode".into(), "aggregate".into(), "--proofs".into(), groups.join(";"), "--parent".into(), parent, "--out".into(), as_host_path(&results)];
if let Some(pf) = prev_file {
args.push("--prev".into());
args.push(pf);
}
let (ok, out) = run_tool(shared, t, &t.host, &args, &[("SP1_PROVER", if t.cuda { "cuda" } else { "cpu" }), ("RUST_LOG", "off")], Duration::from_secs(2 * 3600), &dir.join("aggregate.log"));
if !ok || !results.exists() {
return Err(format!("segment {first}..{last}: aggregator: {}", out.lines().rev().find(|l| l.contains("RESULT") || l.contains("rror")).unwrap_or("failed")));
}
let res: Value = serde_json::from_str(&std::fs::read_to_string(&results).map_err(|e| e.to_string())?).map_err(|e| e.to_string())?;
let pv = res["segment_public_values"].as_str().ok_or("no public values in the results")?.to_string();
let proof_sha = res["segment_proof_sha256"].as_str().ok_or("no proof hash in the results")?.to_string();
let proof_file = res["segment_proof_file"].as_str().ok_or("no proof file in the results")?.to_string();
// the node's native statement, every field but provers (bytes 236..268 of the 340)
let strip = |h: &str| { let h = h.trim_start_matches("0x"); if h.len() == 680 { format!("{}{}", &h[..472], &h[536..]) } else { h.to_string() } };
if strip(&pv) != strip(&expected_pv) {
return Err(format!("segment {first}..{last}: the aggregated statement differs from the node's native statement (it would be vetoed); ours {} node {}", &pv[..66.min(pv.len())], &expected_pv[..66.min(expected_pv.len())]));
}
let proof_path = if t.wsl { PathBuf::from(proof_file.replace("/mnt/c/", "C:/")) } else { PathBuf::from(proof_file) };
let sg = crate::detect::run_timeout(crate::platform::quiet(&mut Command::new(&t.miner)).args(["sign-segment-record", label, &chain_name(shared), &first.to_string(), &last.to_string(), &last_hash, payout, &pv, &proof_sha]), None, Duration::from_secs(20)).ok_or("sign-segment-record did not run")?;
let signed: Value = serde_json::from_str(sg.lines().last().unwrap_or("")).map_err(|_| format!("sign-segment-record: {}", sg.trim()))?;
let record = signed["record"].as_str().ok_or("sign-segment-record gave no record")?.to_string();
let proof = std::fs::read(&proof_path).map_err(|e| format!("proof file {}: {e}", proof_path.display()))?;
let proof_hex = format!("0x{}", proof.iter().map(|b| format!("{b:02x}")).collect::<String>());
let r = evm_rpc(shared, "igneum_submitSegmentRecord", json!([{ "record": record, "proof": proof_hex }]), Duration::from_secs(60))?;
if !r["accepted"].as_bool().unwrap_or(false) {
return Err(format!("segment {first}..{last}: record refused: {}", r["reason"].as_str().unwrap_or("?")));
}
let secs = started.elapsed().as_secs_f64();
shared.event("proving", &format!("segment {first}..{last} aggregated and submitted in {secs:.0} s (chain_len {})", res["segment_chain_len"]));
set(shared, |p| p.aggregated += 1);
Ok(Some(format!("segment {first}..{last} aggregated in {secs:.0} s and submitted; paid when a block carries it")))
}
/// Windows: runs the WSL2 setup from the payload (`wsl2/setup-wsl.sh` next to the engine) in a window of its own;
/// the user watches it and reboots when it asks. Elsewhere there is nothing to set up.
pub fn setup(shared: &Shared) -> Result<Value, String> {
@ -544,6 +1060,16 @@ pub fn setup(shared: &Shared) -> Result<Value, String> {
mod tests {
use super::*;
#[test]
fn pinned_ids_come_out_of_the_hosts_id_line() {
let out = "RESULT id: pinned guests: shard program id 0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a (2832504 bytes, sha256 0x150f4c05a2951fc5) aggregator id 0x474678f35f7545db28055d5e5bbc308231d84a5a072202087a2a8d5b09123896 (319744 bytes), pinned 2026-10-05T16:20:38Z on Darwin, SP1 5.0 circuit v5\n";
let (shard, agg) = ids_from_describe(out).unwrap();
assert_eq!(shard, "0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a");
assert_eq!(agg, "0x474678f35f7545db28055d5e5bbc308231d84a5a072202087a2a8d5b09123896");
assert_eq!(ids_from_describe("RESULT setup: 1.2 s"), None);
assert_eq!(ids_from_describe("shard program id 0xabc aggregator id 0xdef"), None);
}
#[test]
fn chooses_the_newest_unpaid_assigned_shard_not_yet_attempted() {
let v: Value = serde_json::from_str(r#"[

View file

@ -0,0 +1,214 @@
//! Proving v1 (spec 7.8): segment-aligned work for the prover loop (6 October 2026).
//!
//! The shipped loop took the newest open shard each pass, so one prover scattered one block in about 45 across
//! the segment grid and no segment ever had all its blocks proven (node 1, 04:16Z: pending 55, proven 0). Here a
//! free prover claims a whole segment (`proving_v1_segment_blocks` consecutive chain blocks), proves every shard
//! of it in order from one export in one host run (`--mode chain --save-shards`), submits the shard records and the
//! aggregated segment record, then takes the next. One card completes whole segments at its own rate instead of
//! completing none.
//!
//! The choice is deterministic per prover: among the untouched whole segments still inside their deadline by a
//! margin, the lowest FNV-1a of (first block, this prover's key hash) wins, so several provers spread over the
//! candidates without a coordinator; the per-block fallback (`prover::choose`) stays for the passes where no whole
//! segment qualifies.
use std::collections::{BTreeMap, HashSet};
use crate::prover::Work;
/// The least time a claimed segment is given before its deadline (DAA units, about one a second on devnet): the
/// chain of 8 empty blocks took 135.6 s cold beside the miner (bench-log, 5 October 2026), so 240 leaves the
/// submission and the carrying block inside the window.
pub const SEGMENT_MARGIN_MIN_DAA: u64 = 240;
/// The margin grows with what the last segment actually took, times this.
pub const SEGMENT_MARGIN_FACTOR: f64 = 1.5;
/// How far back the work list reaches (chain blocks): the record window, so every open segment inside the
/// deadline is visible.
pub const WORK_LOOKBACK: u64 = 600;
#[derive(Clone, Debug, PartialEq)]
pub struct SegmentWork {
pub first: u64,
pub last: u64,
/// the last block's DAA score; the deadline is it plus `proving_v1_unproven_daa`
pub last_daa: u64,
pub deadline_daa: u64,
/// (chain block number, block hash, shard index) in chain order
pub shards: Vec<(u64, String, u32)>,
/// the shard payouts summed (what the shards earn when carried)
pub shard_wei: u128,
}
/// The segment holding chain block `number` on the grid that starts at `start`.
pub fn segment_of(start: u64, n: u64, number: u64) -> (u64, u64) {
let n = n.max(1);
let k = number.saturating_sub(start) / n;
(start + k * n, start + k * n + n - 1)
}
/// The DAA margin a segment must have before its deadline: the floor, or 1.5 times the last segment's wall time.
pub fn need_daa(last_segment_secs: Option<f64>) -> u64 {
let from_last = last_segment_secs.map(|s| (s * SEGMENT_MARGIN_FACTOR).ceil() as u64).unwrap_or(0);
from_last.max(SEGMENT_MARGIN_MIN_DAA)
}
/// Groups the node's work list into whole, untouched segments: every block of the segment is in the list, every
/// listed shard is open (past its exclusive window), unpaid and not in this node's pool from us. A segment with a
/// block missing (inside the exclusive window, or outside the lookback) or a shard already paid is not a candidate.
pub fn whole_segments(start: u64, n: u64, unproven_daa: u64, work: &[Work]) -> Vec<SegmentWork> {
let n = n.max(1);
let mut by_block: BTreeMap<u64, Vec<&Work>> = BTreeMap::new();
for w in work.iter().filter(|w| w.number >= start) {
by_block.entry(w.number).or_default().push(w);
}
let mut out = Vec::new();
let mut seen = HashSet::new();
for number in by_block.keys() {
let (first, last) = segment_of(start, n, *number);
if !seen.insert(first) {
continue;
}
let mut shards = Vec::new();
let mut wei: u128 = 0;
let mut last_daa = 0;
let mut whole = true;
for b in first..=last {
let Some(entries) = by_block.get(&b) else {
whole = false;
break;
};
let mut e: Vec<&&Work> = entries.iter().collect();
e.sort_by_key(|w| w.shard);
e.dedup_by_key(|w| w.shard);
if e.iter().any(|w| !w.open || w.paid || w.in_pool) {
whole = false;
break;
}
for w in e {
shards.push((w.number, w.hash.clone(), w.shard));
wei = wei.saturating_add(w.shard_wei);
if b == last {
last_daa = w.daa;
}
}
}
if whole && !shards.is_empty() {
out.push(SegmentWork { first, last, last_daa, deadline_daa: last_daa.saturating_add(unproven_daa), shards, shard_wei: wei });
}
}
out
}
/// FNV-1a 64 of the segment's first block and this prover's key hash: the per-prover rank.
pub fn rank(first: u64, key_hash: &str) -> u64 {
let mut h: u64 = 0xcbf29ce484222325;
for b in first.to_be_bytes().iter().chain(key_hash.as_bytes()) {
h ^= *b as u64;
h = h.wrapping_mul(0x100000001b3);
}
h
}
/// The segments to try, best first: inside the deadline by `need` DAA at `tip_daa`, not attempted, ranked by
/// `rank(first, key)` (ties by the older first block).
pub fn candidates(segs: &[SegmentWork], tip_daa: u64, need: u64, key_hash: &str, attempted: &HashSet<u64>) -> Vec<SegmentWork> {
let mut c: Vec<SegmentWork> = segs.iter().filter(|s| !attempted.contains(&s.first) && s.deadline_daa >= tip_daa.saturating_add(1).saturating_add(need)).cloned().collect();
c.sort_by(|a, b| rank(a.first, key_hash).cmp(&rank(b.first, key_hash)).then(a.first.cmp(&b.first)));
c
}
#[cfg(test)]
mod tests {
use super::*;
fn w(number: u64, shard: u32, daa: u64, open: bool, paid: bool, in_pool: bool) -> Work {
Work { number, hash: format!("0x{number:064x}"), shard, pgas: 0, tx_count: 0, assigned: false, open, paid, in_pool, shard_wei: 10, key_hash: String::new(), daa }
}
fn grid(start: u64, n: u64, segments: u64, daa0: u64) -> Vec<Work> {
(0..segments * n).map(|i| w(start + i, 0, daa0 + i, true, false, false)).collect()
}
#[test]
fn the_grid_is_counted_from_the_first_v1_block() {
assert_eq!(segment_of(100, 8, 100), (100, 107));
assert_eq!(segment_of(100, 8, 107), (100, 107));
assert_eq!(segment_of(100, 8, 108), (108, 115));
assert_eq!(segment_of(100, 8, 123), (116, 123));
}
#[test]
fn only_whole_open_unpaid_untouched_segments_qualify() {
let mut work = grid(100, 4, 3, 1000); // 100..111, three segments
work.retain(|x| x.number != 105); // 104..107 has a block missing (inside its exclusive window, say)
work.iter_mut().find(|x| x.number == 110).unwrap().paid = true; // 108..111 has a paid shard
let segs = whole_segments(100, 4, 600, &work);
assert_eq!(segs.len(), 1);
assert_eq!((segs[0].first, segs[0].last), (100, 103));
assert_eq!(segs[0].shards.len(), 4);
assert_eq!(segs[0].last_daa, 1003);
assert_eq!(segs[0].deadline_daa, 1603);
assert_eq!(segs[0].shard_wei, 40);
// a shard of ours already in the pool, or one still exclusive, also disqualifies
let mut work = grid(100, 4, 1, 1000);
work[1].in_pool = true;
assert!(whole_segments(100, 4, 600, &work).is_empty());
let mut work = grid(100, 4, 1, 1000);
work[3].open = false;
assert!(whole_segments(100, 4, 600, &work).is_empty());
}
#[test]
fn a_block_with_several_shards_lists_them_in_order() {
let mut work = grid(100, 2, 1, 1000);
work.push(w(101, 1, 1001, true, false, false));
work.push(w(100, 1, 1000, true, false, false));
let segs = whole_segments(100, 2, 600, &work);
assert_eq!(segs[0].shards.iter().map(|(n, _, s)| (*n, *s)).collect::<Vec<_>>(), vec![(100, 0), (100, 1), (101, 0), (101, 1)]);
}
#[test]
fn the_deadline_margin_and_the_attempted_set_filter_the_candidates() {
let work = grid(100, 8, 4, 1000); // 100..131, deadlines 1607, 1615, 1623, 1631
let segs = whole_segments(100, 8, 600, &work);
assert_eq!(segs.len(), 4);
// at tip DAA 1380 with a 240 margin only the segments with a deadline at or past 1621 remain
let c = candidates(&segs, 1380, 240, "0xkey", &HashSet::new());
let firsts: Vec<u64> = c.iter().map(|s| s.first).collect();
assert_eq!(firsts.len(), 2);
assert!(firsts.contains(&116) && firsts.contains(&124));
let mut attempted = HashSet::new();
attempted.insert(firsts[0]);
let c2 = candidates(&segs, 1380, 240, "0xkey", &attempted);
assert_eq!(c2.len(), 1);
assert_eq!(c2[0].first, firsts[1]);
// past every deadline: nothing
assert!(candidates(&segs, 1700, 240, "0xkey", &HashSet::new()).is_empty());
}
#[test]
fn the_order_is_deterministic_per_key_and_differs_between_keys() {
let work = grid(100, 8, 6, 1000);
let segs = whole_segments(100, 8, 600, &work);
let a = candidates(&segs, 1000, 240, "0xaaaa", &HashSet::new());
let a2 = candidates(&segs, 1000, 240, "0xaaaa", &HashSet::new());
assert_eq!(a, a2);
assert_eq!(a.len(), 6);
// two provers rank the six candidates differently (the spread); the sets are the same
let b = candidates(&segs, 1000, 240, "0xbbbb", &HashSet::new());
let (fa, fb): (Vec<u64>, Vec<u64>) = (a.iter().map(|s| s.first).collect(), b.iter().map(|s| s.first).collect());
let mut sa = fa.clone();
let mut sb = fb.clone();
sa.sort();
sb.sort();
assert_eq!(sa, sb);
assert_ne!(fa, fb, "two keys should not rank six segments identically");
}
#[test]
fn the_margin_follows_the_last_segment_time() {
assert_eq!(need_daa(None), 240);
assert_eq!(need_daa(Some(100.0)), 240);
assert_eq!(need_daa(Some(190.0)), 285);
}
}

View file

@ -319,6 +319,12 @@ fn api_post(shared: &Arc<Shared>, path: &str, body: Value) -> Result<Value, Stri
shared.send(Cmd::SweepPin(key, pinned));
Ok(json!({ "ok": true }))
}
// Power control (config.rs power_control): on = one administrator prompt now for the cap, off = nothing asks
"/api/power/control" => {
let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?;
shared.send(Cmd::PowerControl(on));
Ok(json!({ "ok": true }))
}
"/api/sweep/enable" => {
let on = body.get("on").and_then(|v| v.as_bool()).ok_or("on missing")?;
shared.send(Cmd::SweepEnable(on));
@ -333,7 +339,8 @@ fn api_post(shared: &Arc<Shared>, path: &str, body: Value) -> Result<Value, Stri
Ok(json!({ "ok": true }))
}
"/api/quit" => {
shared.send(Cmd::Quit);
// the caller is on 127.0.0.1 and holds the token: the installer, the OTA apply, a script that read app.url
shared.send(Cmd::Quit("POST /api/quit (a local caller with the token: the installer, the OTA apply, or a script that read app.url)"));
Ok(json!({ "ok": true }))
}
_ => Err("unknown api".into()),

View file

@ -24,12 +24,26 @@ pub struct NodeState {
/// a consensus switch the signed manifest announced (difficulty v2 activation DAA); 0 = none
pub consensus_switch_daa: u64,
pub override_restart_wait: String,
/// the node's "Consensus params digest: <64 hex>" line (exchanged in the p2p handshake); empty until the node prints it
pub consensus_digest: String,
/// every activation height in the override file the node was started with (miner-ui-2): the Node page names the next one
pub consensus_switches: Vec<ConsensusSwitch>,
}
#[derive(Clone, Serialize, Default)]
/// One planned rule change the node applies by itself at a DAA score (from the override file's `*_activation_daa` keys).
#[derive(Clone, Serialize, Default, PartialEq, Debug)]
pub struct ConsensusSwitch {
pub key: String, // the override key, for example fees_v1_activation_daa
pub name: String, // plain words: Fees v1
pub daa: u64,
}
#[derive(Clone, Serialize, Default, Debug)]
pub struct CardState {
pub index: usize,
pub key: String, // stable id for the saved preference: vendor:device:name
pub key: String, // stable id for the saved preference: vendor:code, "#2" and up for a second identical card (no index: it moves)
pub code: String, // the tool's own device name (AMD's OpenCL runtime says "gfx1201"); the key and the tooltip carry it
pub platform: String, // the OpenCL platform and driver the worker opens it through (the tooltip)
pub kind: String, // apple | discrete | integrated | external | unknown
pub vram_mb: u64, // 0 when unknown
pub reason: String, // why it is off by default, if it is
@ -40,7 +54,7 @@ pub struct CardState {
pub detail: String, // memory, cores
pub device: String, // the worker's --device value (Windows)
pub enabled: bool,
pub state: String, // off | waiting | starting | ready | mining | restarting | failed | faulted (the watchdog gave up on it)
pub state: String, // off | waiting | starting | ready | mining | restarting | failed | faulted (the watchdog gave up on it) | unusable (the OS reports a problem) | removed (unplugged)
pub hash_now: f64, // MH/s, the last interval
pub hash_avg: f64, // MH/s since the start
pub accepted: u64,
@ -57,6 +71,15 @@ pub struct CardState {
pub pid: u32,
pub last_status_age_s: f64,
pub message: String,
/// the device's own problem as the OS reports it ("Code 43" on Windows); set = listed but no worker can drive it
pub problem: String,
/// PCI bus id where the tool gives one (nvidia-smi pci.bus_id); a second way to recognise a card whose index moved
pub bus: String,
/// hot-plug (src/hotplug.rs): unix s when a re-detection added this card (0 = found at start), when it lost it
/// (0 = present), and `gone` once a removed card has been shown as removed for five minutes (the row hides)
pub added_at: f64,
pub removed_at: f64,
pub gone: bool,
/// `WORKER FAULT` lines seen (the miner killed and restarted its worker) and the miner's own `faults=` count
pub faults: u64,
/// shares that failed the miner's CPU re-check this run (`mismatched=` on the STATUS line)
@ -74,6 +97,11 @@ pub struct CardState {
pub temp_gpu: f64,
pub temp_mem: f64,
pub telemetry_at: f64,
// AMD through igneum-gpu-telemetry (ADLX on Windows, amdgpu sysfs on Linux), 5 October 2026; 0 = unknown
pub fan_pct: f64,
pub fan_rpm: f64,
pub mclk_mhz: f64,
pub util_pct: f64,
// hash per watt (src/sweep.rs)
pub eff_mhw: f64, // live: hash_now over power_w, MH per watt; 0 = unknown
pub sweep_supported: bool, // NVIDIA with readable limits; the note says why not otherwise
@ -85,6 +113,18 @@ pub struct CardState {
pub sweep_mhs: f64,
pub sweep_at: f64, // unix s of the last sweep
pub pinned: bool, // the user set the cap by hand; the sweep records but does not change it
// Ember Tune (src/ember.rs): the two-knob tune, 5 October 2026
pub clock_max_mhz: u32, // the vendor's maximum core clock (0 = unknown)
pub clock_min_mhz: u32, // the vendor's floor for a cap (0 = 60% of the maximum)
pub gclk_mhz: f64, // core clock now
pub clock_cap_mhz: u32, // the cap in force (0 = unlocked)
pub amd_ordinal: i64, // the `amd N` ordinal of igneum-gpu-telemetry (-1 = unknown)
pub driver: String, // the driver version (nvidia-smi, or the worker's race line)
pub program_class: String, // the program class of the race line (loads and wide loads per hash); "" = unknown
pub tune_control: bool, // both knobs reach the card (else measure only; sweep_note says why)
pub tune_clock_mhz: u32, // the clock cap the last tune chose (0 = unlocked)
pub tune_source: String, // full | confirm | baseline
pub tune_line: String, // "Tuned: 122.3 MH/s at 290 W (0.422 MH/W)" once tuned
// the kernel variant race (docs/design/miner-tuning.md): what the worker's last race chose
pub variant: String,
pub race_mhs: f64,
@ -92,6 +132,14 @@ pub struct CardState {
pub race_variants: u32,
}
impl CardState {
/// Listed, driver fine, not unplugged: a worker can run on it (hot-plug keeps removed and faulty cards in the
/// list so the other cards' indices stay put).
pub fn present(&self) -> bool {
self.removed_at == 0.0 && self.problem.is_empty()
}
}
#[derive(Clone, Serialize, Default)]
pub struct MiningState {
pub state: String, // idle | waiting | mining | paused | stopped
@ -141,6 +189,9 @@ pub struct ProvingState {
pub submitted: u32,
pub paid: u32,
pub failed: u32,
/// wei, serialised as a decimal string: serde_json's `to_value` refuses a u128 over u64::MAX (about 18.45 IGN,
/// 15 paid shards at 1.23 IGN), and that refusal emptied the whole `/api/state` reply to "{}" (6 October 2026)
#[serde(serialize_with = "u128_string")]
pub paid_wei: u128,
pub current: String,
pub started_at: f64,
@ -162,6 +213,23 @@ pub struct ProvingState {
pub pool_entries: u64,
pub pool_verified: u64,
pub pool_failed: u64,
/// the pinned guests' ids (`igneum-prove-host --mode id`): the shard program and the aggregator; empty until read
pub program_id: String,
pub aggregator_id: String,
/// proving v1 step 1: the install-time default's one plain line (why proving is on or off on this machine)
pub default_note: String,
/// proving v1: segment records this machine aggregated and submitted, and the aggregator's last line
pub aggregated: u32,
pub segment_note: String,
/// proving v1 segment path (6 October 2026): whole segments this machine proved and submitted, paid, and what
/// they paid (wei as a decimal string, see `paid_wei`); the last segment's wall time
pub segments_submitted: u32,
pub segments_paid: u32,
#[serde(serialize_with = "u128_string")]
pub segment_paid_wei: u128,
pub segment_last_s: f64,
/// segment records the node refused by the chain rule and this machine offers again each pass
pub segments_held: u32,
}
#[derive(Clone, Serialize, Default)]
@ -205,8 +273,16 @@ pub struct SettingsState {
pub remote_jobs: bool,
/// the prover service (src/prover.rs)
pub prove: bool,
/// the efficiency sweep (src/sweep.rs): once after install, then weekly
/// the efficiency sweep (src/sweep.rs): once after install, then weekly; effective only with `power_control`
pub sweep: bool,
/// the NVIDIA power cap and the sweep may ask for administrator rights (config.rs: default off, one prompt when
/// switched on)
pub power_control: bool,
/// the line beside the Power control switch: why it is off, or that the rights were given
pub power_note: String,
/// Ember Tune is paused fleet-wide by the signed manifest's kill switch (tuning.ember.enabled = false)
pub tuning_off: bool,
pub tuning_note: String,
/// the miner software's dev fee switch (settings; `--dev-fee 0` when off)
pub dev_fee: bool,
/// devnet only: the node trusts proof records without a verifier (`IGNEUM_PROOF_VERIFY=trust`)
@ -362,3 +438,22 @@ impl Rings {
out
}
}
/// A u128 as a decimal JSON string (the dashboard reads it with `Number()`).
pub fn u128_string<S: serde::Serializer>(v: &u128, s: S) -> Result<S::Ok, S::Error> {
s.serialize_str(&v.to_string())
}
#[cfg(test)]
mod paid_wei_tests {
use super::*;
#[test]
fn a_paid_total_over_u64_max_still_serialises_the_whole_state() {
let mut st = State::default();
st.proving.paid_wei = u64::MAX as u128 + 1;
let v = serde_json::to_value(&st).expect("the state serialises");
assert_eq!(v["proving"]["paid_wei"], serde_json::Value::String("18446744073709551616".into()));
assert!(v["mining"].is_object());
}
}

View file

@ -340,10 +340,12 @@ impl Run {
}
}
/// The elevated helper that sets caps for a sweep (one administrator prompt per sweep, not one per step). It polls
/// `<dir>/cmd.txt` twice a second: a line `<seq> <watts>` runs `nvidia-smi -i <device> -pl <watts>`, `quit` ends it.
/// After 20 minutes without a new command it restores `<restore watts>` and exits by itself, so an engine that died
/// mid-sweep leaves the card on its old limit. It writes what it ran to `<dir>/helper.log`.
/// The elevated helper that sets limits for a tune (one administrator prompt per tune, not one per step). It polls
/// `<dir>/cmd.txt` twice a second; each line is `<seq> pl <watts>` (`nvidia-smi -i <device> -pl <watts>`; the
/// 0.3.9 form `<seq> <watts>` still works), `<seq> lgc <mhz>` (`-lgc 0,<mhz>`, the core clock cap; the memory clock
/// is never touched) or `<seq> rgc` (`-rgc`, unlocked); `quit` ends it. After 20 minutes without a new command it
/// restores `<restore watts>`, resets the clocks and exits by itself, so an engine that died mid-tune leaves the
/// card on its old limits. It writes what it ran to `<dir>/helper.log`.
pub fn helper_script_windows() -> &'static str {
r#"param([string]$Dir, [string]$Smi, [string]$Device, [string]$Restore)
$ErrorActionPreference = 'Continue'
@ -359,15 +361,27 @@ while ($true) {
$last = $c
$idle = Get-Date
if ($c -eq 'quit') { "$(Get-Date -Format o) quit" | Out-File -FilePath $log -Append -Encoding utf8; break }
$w = ($c -split ' ')[-1]
if ($w -match '^\d+$') {
$out = (& $Smi -i $Device -pl $w 2>&1 | Out-String).Trim()
"$(Get-Date -Format o) -pl $w : $out" | Out-File -FilePath $log -Append -Encoding utf8
foreach ($line in ($c -split "`n")) {
$p = ($line.Trim() -split ' ')
if ($p.Count -lt 2) { continue }
$op = $p[1]; $v = $p[-1]
if ($p.Count -eq 2 -and $v -match '^\d+$') { $op = 'pl' }
if ($op -eq 'pl' -and $v -match '^\d+$') {
$out = (& $Smi -i $Device -pl $v 2>&1 | Out-String).Trim()
"$(Get-Date -Format o) $($p[0]) -pl $v : $out" | Out-File -FilePath $log -Append -Encoding utf8
} elseif ($op -eq 'lgc' -and $v -match '^\d+$') {
$out = (& $Smi -i $Device -lgc "0,$v" 2>&1 | Out-String).Trim()
"$(Get-Date -Format o) $($p[0]) -lgc 0,$v : $out" | Out-File -FilePath $log -Append -Encoding utf8
} elseif ($op -eq 'rgc') {
$out = (& $Smi -i $Device -rgc 2>&1 | Out-String).Trim()
"$(Get-Date -Format o) $($p[0]) -rgc : $out" | Out-File -FilePath $log -Append -Encoding utf8
}
}
}
if (((Get-Date) - $idle).TotalMinutes -gt 20) {
$out = (& $Smi -i $Device -pl $Restore 2>&1 | Out-String).Trim()
"$(Get-Date -Format o) idle 20 min: restored $Restore W and quit: $out" | Out-File -FilePath $log -Append -Encoding utf8
$out2 = (& $Smi -i $Device -rgc 2>&1 | Out-String).Trim()
"$(Get-Date -Format o) idle 20 min: restored $Restore W, clocks reset, and quit: $out / $out2" | Out-File -FilePath $log -Append -Encoding utf8
break
}
Start-Sleep -Milliseconds 500
@ -388,11 +402,20 @@ while true; do
if [ -n "$c" ] && [ "$c" != "$last" ]; then
last="$c"; idle=$(date +%s)
if [ "$c" = "quit" ]; then echo "$(date -u +%FT%TZ) quit" >> "$dir/helper.log"; break; fi
w="${c##* }"
case "$w" in ''|*[!0-9]*) ;; *) echo "$(date -u +%FT%TZ) -pl $w : $("$smi" -i "$dev" -pl "$w" 2>&1)" >> "$dir/helper.log";; esac
printf '%s\n' "$c" | while IFS= read -r line; do
set -- $line
[ $# -ge 2 ] || continue
op="$2"; v="${line##* }"
[ $# -eq 2 ] && op=pl
case "$op" in
pl) case "$v" in ''|*[!0-9]*) ;; *) echo "$(date -u +%FT%TZ) $1 -pl $v : $("$smi" -i "$dev" -pl "$v" 2>&1)" >> "$dir/helper.log";; esac ;;
lgc) case "$v" in ''|*[!0-9]*) ;; *) echo "$(date -u +%FT%TZ) $1 -lgc 0,$v : $("$smi" -i "$dev" -lgc "0,$v" 2>&1)" >> "$dir/helper.log";; esac ;;
rgc) echo "$(date -u +%FT%TZ) $1 -rgc : $("$smi" -i "$dev" -rgc 2>&1)" >> "$dir/helper.log" ;;
esac
done
fi
if [ $(( $(date +%s) - idle )) -gt 1200 ]; then
echo "$(date -u +%FT%TZ) idle 20 min: restored $restore W: $("$smi" -i "$dev" -pl "$restore" 2>&1)" >> "$dir/helper.log"; break
echo "$(date -u +%FT%TZ) idle 20 min: restored $restore W, clocks reset: $("$smi" -i "$dev" -pl "$restore" 2>&1) / $("$smi" -i "$dev" -rgc 2>&1)" >> "$dir/helper.log"; break
fi
sleep 0.5
done
@ -405,8 +428,9 @@ pub fn unsupported_reason(vendor: &str, power_default_w: f64, device: &str) -> O
match vendor {
"nvidia" if power_default_w > 0.0 && !device.is_empty() => None,
"nvidia" => Some("not available: nvidia-smi did not report this card's power limits"),
"apple" => Some("not available on Apple silicon: there is no power cap to set, and powermetrics needs administrator rights for the draw"),
"amd" => Some("not available for AMD in this version: the app has no power reading or cap for AMD cards (nothing like nvidia-smi ships with the driver)"),
// Ember Tune (src/ember.rs, 5 October 2026): AMD is tuned through igneum-gpu-telemetry, Apple measures only;
// the tune itself says which at its start (the card row's note)
"apple" | "amd" => None,
_ => Some("not available: no power reading or cap for this card"),
}
}
@ -580,14 +604,16 @@ mod tests {
fn unsupported_reasons() {
assert!(unsupported_reason("nvidia", 575.0, "0").is_none());
assert!(unsupported_reason("nvidia", 0.0, "0").unwrap().contains("power limits"));
assert!(unsupported_reason("apple", 0.0, "").unwrap().contains("powermetrics"));
assert!(unsupported_reason("amd", 0.0, "1").unwrap().contains("AMD"));
assert!(unsupported_reason("apple", 0.0, "").is_none(), "measure only, said by the tune");
assert!(unsupported_reason("amd", 0.0, "1").is_none(), "tuned through igneum-gpu-telemetry");
assert!(unsupported_reason("other", 0.0, "1").is_some());
}
#[test]
fn helper_scripts_carry_the_protocol() {
for s in [helper_script_windows(), helper_script_unix()] {
assert!(s.contains("cmd.txt") && s.contains("quit") && s.contains("-pl") && s.contains("20 min"));
assert!(s.contains("-lgc") && s.contains("-rgc"), "the clock cap and its reset");
}
}
}

View file

@ -27,6 +27,12 @@ pub const HEALTHY_RESET_S: f64 = 300.0;
pub const MINER_GAVE_UP_CODE: i32 = 43;
/// No `watch` reading from our node for this long: restart it.
pub const NODE_SILENT_S: f64 = 120.0;
/// `igneum-miner` exit code when its worker refused the program pack it was started with and the miner could not
/// rebuild it (or rebuilt it `PACK_REBUILD_CAP` times this epoch): the app exports the pack from the node again
/// before the next start, at most `PACK_REBUILD_CAP` times per epoch, then shows the card.
pub const PACK_OUT_OF_DATE_CODE: i32 = 44;
/// Pack rebuilds per epoch seed before the app stops restarting the miner on it.
pub const PACK_REBUILD_CAP: u32 = 3;
/// Node restarts inside this window grow the delay before the next one.
pub const NODE_WINDOW_S: f64 = 600.0;
@ -280,10 +286,114 @@ fn kv_f64(line: &str, key: &str) -> Option<f64> {
kv(line, key)?.trim_end_matches('s').parse().ok()
}
/// The decision after a pack refusal (exit `PACK_OUT_OF_DATE_CODE`, or a `PACK OUT OF DATE` / `error 0 pack` line
/// from the miner): rebuild the pack before the restart, or stop for this epoch. 5 October 2026: the app restarted
/// two workers every 20 to 40 s for an hour on a pack neither it nor the miner had re-exported in between.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum PackAction {
/// Export the pack from the node again (the n-th time this epoch) and start the miner on it now.
Rebuild { n: u32, cap: u32 },
/// The cap for this epoch is reached: show the card with this reason and wait for the next epoch (or a reset).
GiveUp { n: u32, reason: String },
}
/// Pack rebuilds per epoch seed (the epoch the exported pack names in its seeds.txt).
#[derive(Debug, Default)]
pub struct PackRebuilds {
epoch: Option<String>,
n: u32,
}
impl PackRebuilds {
pub fn new() -> Self {
Self::default()
}
/// The miner exited over its pack while the exported pack is for `epoch`; `why` is the miner's reason.
pub fn decide(&mut self, epoch: &str, why: &str) -> PackAction {
if self.epoch.as_deref() != Some(epoch) {
self.epoch = Some(epoch.to_string());
self.n = 0;
}
if self.n >= PACK_REBUILD_CAP {
return PackAction::GiveUp { n: self.n, reason: format!("the program pack still fails after {} rebuilds this epoch ({why}); the worker and the pack disagree", self.n) };
}
self.n += 1;
PackAction::Rebuild { n: self.n, cap: PACK_REBUILD_CAP }
}
pub fn count(&self) -> u32 {
self.n
}
}
/// A miner line that names a pack refusal: the worker's own `error 0 pack <dir>: <why>` (relayed as
/// `worker error: ...`) or the miner's `PACK OUT OF DATE <dir>: <why>`. Returns the reason, in plain words.
pub fn pack_refusal(text: &str) -> Option<String> {
if let Some(i) = text.find("PACK OUT OF DATE") {
let rest = text[i + "PACK OUT OF DATE".len()..].trim_start_matches(':').trim();
return Some(rest.split_once(": ").map(|(_, why)| why).unwrap_or(rest).to_string());
}
if let Some(i) = text.find("error 0 pack ") {
let rest = &text[i + "error 0 pack ".len()..];
let (_, why) = rest.split_once(": ")?;
return Some(why.trim().to_string());
}
None
}
/// The epoch seed hex an exported pack names (`epoch_seed_hex <hex>` in its seeds.txt), 16 chars.
pub fn pack_epoch_of(seeds_txt: &str) -> Option<String> {
seeds_txt.lines().find_map(|l| l.strip_prefix("epoch_seed_hex ")).map(|h| h.trim().chars().take(16).collect())
}
#[cfg(test)]
mod tests {
use super::*;
// The refusal as PC 1 and PC 2 logged it on 5 October 2026 (epoch 34), the miner's own line, and non-refusals.
const REFUSAL: &str = "! 1791224840.037 worker error: error 0 pack packs\\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT (wrong seeds.txt for this pack?)";
const MINER_LINE: &str = "1791224840.100 PACK OUT OF DATE packs\\devnet: the worker refused its program pack (the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT); rebuilding the program pack before the restart";
#[test]
fn pack_refusal_reads_both_lines_and_nothing_else() {
assert_eq!(pack_refusal(REFUSAL).unwrap(), "the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT (wrong seeds.txt for this pack?)");
assert!(pack_refusal(MINER_LINE).unwrap().starts_with("the worker refused its program pack"));
assert_eq!(pack_refusal("! 1791224651.247 worker error: error 5838 epoch seed mismatch: this worker holds epoch bed7ab62cbece66c"), None);
assert_eq!(pack_refusal(STATUS_OK), None);
assert_eq!(pack_epoch_of("epoch_seed_hex 009858237e118f69abc8d096e9b1af21c24539eaecdfd1b896588825660a69ec\nday_seed_hex 69676e65\n").as_deref(), Some("009858237e118f69"));
assert_eq!(pack_epoch_of("day_seed_hex 69676e65\n"), None);
}
/// Known-good: a refusal rebuilds the pack, once per refusal, and a new epoch starts the count again.
#[test]
fn pack_rebuilds_known_good() {
let mut p = PackRebuilds::new();
assert_eq!(p.decide("009858237e118f69", "x"), PackAction::Rebuild { n: 1, cap: PACK_REBUILD_CAP });
assert_eq!(p.decide("009858237e118f69", "x"), PackAction::Rebuild { n: 2, cap: PACK_REBUILD_CAP });
assert_eq!(p.decide("5e5d0c3b2a19f8e7", "x"), PackAction::Rebuild { n: 1, cap: PACK_REBUILD_CAP });
assert_eq!(p.count(), 1);
}
/// Known-mismatched: a pack the worker refuses after every rebuild stops at the cap with the reason in plain
/// words, and stays stopped for that epoch.
#[test]
fn pack_rebuilds_known_mismatched_gives_up_at_the_cap() {
let mut p = PackRebuilds::new();
for n in 1..=PACK_REBUILD_CAP {
assert_eq!(p.decide("009858237e118f69", "the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT"), PackAction::Rebuild { n, cap: PACK_REBUILD_CAP });
}
match p.decide("009858237e118f69", "the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") {
PackAction::GiveUp { n, reason } => {
assert_eq!(n, PACK_REBUILD_CAP);
assert_eq!(reason, "the program pack still fails after 3 rebuilds this epoch (the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT); the worker and the pack disagree");
}
a => panic!("{a:?}"),
}
assert!(matches!(p.decide("009858237e118f69", "x"), PackAction::GiveUp { .. }));
assert_eq!(p.decide("5e5d0c3b2a19f8e7", "x"), PackAction::Rebuild { n: 1, cap: PACK_REBUILD_CAP });
}
// Lines as igneum-miner 0.3.x prints them (devnet v4, 4 October 2026; identities and labels shortened).
const STATUS_OK: &str = "1791138616.597 STATUS 'win-1' [worker]: 70s jobs=486 accepted=3 rejected=0 mismatched=0 extra=0 rate=0.04 blocks/s hash=123.90 MH/s wall (124.20 MH/s inside jobs) now=124.10 MH/s wall (124.30 MH/s inside jobs, 70 jobs, seed walk 0 calls) template_age=0.31s synced=true idle=0.2% (last 10s: 0.1%) queued=2 restarts=0 faults=0 identities=8 accepted_by_identity=1/0/1/0/0/1/0/0";
const STATUS_ZERO: &str = "1791138626.597 STATUS 'win-1' [worker]: 80s jobs=486 accepted=3 rejected=0 mismatched=0 extra=0 rate=0.04 blocks/s hash=108.41 MH/s wall (124.20 MH/s inside jobs) now=0.00 MH/s wall (0.00 MH/s inside jobs, 0 jobs, seed walk 0 calls) template_age=0.31s synced=true idle=12.5% (last 10s: 100.0%) queued=2 restarts=0 faults=0 identities=8 accepted_by_identity=1/0/1/0/0/1/0/0";

View file

@ -41,6 +41,43 @@ pub fn candidates(bin_dir: &Path) -> Vec<String> {
}
/// The candidates as one line for a message.
/// Whether the distribution answers at all (`wsl.exe -d Ubuntu-24.04 -- echo <marker>` within 30 s): the install-time
/// prover default (src/provedefault.rs) needs WSL2 on Windows before it switches proving on. Elsewhere: false.
#[allow(dead_code)] // also compiled into src/bin/prove-verify.rs, which does not call it
pub fn distro_answers() -> bool {
if !cfg!(windows) {
return false;
}
// self-contained (this file is also compiled into src/bin/prove-verify.rs, which has no detect or platform module)
let mut cmd = std::process::Command::new("wsl");
cmd.args(["-d", DISTRO, "--", "echo", "igneum-wsl-answers"]).stdin(std::process::Stdio::null()).stdout(std::process::Stdio::piped()).stderr(std::process::Stdio::null());
#[cfg(windows)]
{
use std::os::windows::process::CommandExt;
cmd.creation_flags(0x0800_0000); // CREATE_NO_WINDOW
}
let Ok(mut child) = cmd.spawn() else { return false };
let Some(out) = child.stdout.take() else { return false };
let reader = std::thread::spawn(move || {
let mut s = String::new();
let _ = std::io::Read::read_to_string(&mut std::io::BufReader::new(out), &mut s);
s
});
let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30);
loop {
match child.try_wait() {
Ok(Some(_)) => break,
Ok(None) if std::time::Instant::now() < deadline => std::thread::sleep(std::time::Duration::from_millis(100)),
_ => {
let _ = child.kill();
let _ = child.wait();
break;
}
}
}
reader.join().map(|o| o.replace('\0', "").contains("igneum-wsl-answers")).unwrap_or(false)
}
pub fn candidates_text(bin_dir: &Path) -> String {
candidates(bin_dir).join(", ")
}
@ -152,6 +189,7 @@ pub fn bash_line(file: &Path, login: bool, args: &[&str]) -> String {
/// command line exactly as `bash_line` wrote it; elsewhere the words are ordinary arguments (nothing runs wsl there).
/// The caller adds stdio, the hidden-window flag and the timeout.
pub fn command(wsl_exe: &Path, distro: &str, user: Option<&str>, file: &Path, login: bool, args: &[&str]) -> Command {
// console: a builder; every caller runs it through run_capture, run_streamed or platform::quiet (tools/ci/windows-spawn-check.mjs)
let mut c = Command::new(wsl_exe);
c.args(["-d", distro]);
if let Some(u) = user.filter(|u| !u.is_empty()) {

View file

@ -1,6 +1,9 @@
/* Igneum Miner dashboard. The site's tokens (site/index.html): obsidian, graphite, ember, molten, bone, ash;
Unbounded for headings, IBM Plex Sans for text, IBM Plex Mono for numbers and labels. Fonts ship in the binary.
One type scale (--t-*), one spacing scale (--s-*), one colour set (:root), used by every screen. */
/* Igneum Miner dashboard (miner-ui-2). The site's tokens (site/index.html): obsidian, graphite, ember, molten, bone,
ash; Unbounded for headings, IBM Plex Sans for text, IBM Plex Mono for numbers and labels. Fonts ship in the binary.
One type scale (--t-*), one spacing scale (--s-*), one colour set (:root), one accent (ember), used by every page.
Layout: a rail on the left (six sections), a thin top bar, the status strip under it, the page in the middle, the
log drawer at the bottom. The window lays out from 900 x 600 up (the hosts say the same minimum).
Nothing here is newer than 2022 CSS (the WebView2 host). */
@font-face{font-family:'IBM Plex Mono';font-style:normal;font-weight:400;font-display:swap;src:url(fonts/IBMPlexMono-400.woff2) format('woff2')}
@font-face{font-family:'IBM Plex Mono';font-style:normal;font-weight:500;font-display:swap;src:url(fonts/IBMPlexMono-500.woff2) format('woff2')}
@font-face{font-family:'IBM Plex Sans';font-style:normal;font-weight:400;font-display:swap;src:url(fonts/IBMPlexSans-400.woff2) format('woff2')}
@ -13,14 +16,14 @@
:root{
/* colour */
--obsidian:#0C0C0E;--graphite:#16161A;--line:#2A2A30;--line-2:#3A3A42;--ember:#F2541B;--ember-hi:#FF6A2B;--molten:#FFB35C;--bone:#F4F1EC;--ash:#9A9A9E;--ink-2:#C9C7C2;--ember-ink:#0C0C0E;
--ember-12:rgba(242,84,27,.12);--ember-40:rgba(242,84,27,.4);--molten-10:rgba(255,179,92,.1);--molten-40:rgba(255,179,92,.4);--node-blue:#7FA7C9;--nvidia:#8BE37A;
--ember-12:rgba(242,84,27,.12);--ember-40:rgba(242,84,27,.4);--molten-10:rgba(255,179,92,.1);--molten-40:rgba(255,179,92,.4);--node-blue:#7FA7C9;--nvidia:#8BE37A;--rail-bg:#111114;
/* type scale */
--t-xs:11px;--t-sm:12px;--t-base:13px;--t-md:14px;--t-lg:15px;--t-xl:16px;--t-2xl:18px;--t-num:26px;--t-h3:16px;--t-h2:32px;--t-h1:52px;
/* spacing scale */
--s-1:4px;--s-2:8px;--s-3:12px;--s-4:16px;--s-5:20px;--s-6:28px;--s-7:40px;
--gutter:var(--s-6);--card-pad:22px;--card-r:18px;--tile-pad:18px 20px;--tile-r:14px;--gap:var(--s-5);--gap-tile:var(--s-3);
--sans:'IBM Plex Sans',system-ui,-apple-system,sans-serif;--mono:'IBM Plex Mono',ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;--head:'Unbounded',sans-serif;
--top:60px;--bottom:52px;--drawer-h:260px}
--top:60px;--bottom:0px;--drawer-h:260px;--rail:196px}
*{box-sizing:border-box}
html,body{height:100%}
body{margin:0;background:var(--obsidian);color:var(--bone);font-family:var(--sans);font-size:var(--t-lg);line-height:1.5;-webkit-font-smoothing:antialiased;overflow:hidden;user-select:none;-webkit-user-select:none;font-variant-numeric:tabular-nums}
@ -53,21 +56,48 @@ input,textarea{font-variant-numeric:tabular-nums}
.btn.ghost:hover{border-color:var(--line-2);color:var(--bone)}
.btn.ghost.on{color:var(--molten);border-color:var(--molten-40);background:var(--molten-10)}
.btn.danger:hover{color:var(--ember);border-color:var(--ember)}
.btn .chev{transition:transform .2s ease;color:var(--ash)}
.btn.on .chev{transform:rotate(180deg);color:var(--molten)}
.icon-btn{width:38px;height:38px;border-radius:10px;border:1px solid var(--line-2);background:transparent;display:inline-flex;align-items:center;justify-content:center;cursor:pointer;color:var(--ink-2);font-size:20px;line-height:1}
.icon-btn:hover{color:var(--bone);border-color:var(--ash)}
:focus-visible{outline:2px solid var(--ember);outline-offset:3px;border-radius:6px}
@media (prefers-reduced-motion:reduce){.btn,.btn .chev{transition:none}}
@media (prefers-reduced-motion:reduce){.btn{transition:none}}
/* the rail */
.rail{position:fixed;top:0;left:0;bottom:0;width:var(--rail);background:var(--rail-bg);border-right:1px solid var(--line);display:flex;flex-direction:column;z-index:22;padding:14px 12px 14px;-webkit-app-region:drag}
.rail button{-webkit-app-region:no-drag}
.rail-brand{display:flex;align-items:center;gap:10px;padding:6px 10px 18px;min-width:0}
body.mac .rail-brand{padding-top:30px}
.rail-brand .word{font-family:var(--head);font-weight:900;font-size:17px;letter-spacing:.06em}
.rail-nav{display:flex;flex-direction:column;gap:3px}
.nav{position:relative;display:flex;align-items:center;gap:12px;width:100%;min-height:42px;padding:8px 12px;border:1px solid transparent;border-radius:11px;background:transparent;color:var(--ink-2);font-weight:500;font-size:var(--t-md);text-align:left;cursor:pointer;transition:background .15s ease,color .15s ease,border-color .15s ease}
.nav svg{width:19px;height:19px;flex:0 0 19px;fill:none;stroke:currentColor;stroke-width:1.9;stroke-linecap:round;stroke-linejoin:round;color:var(--ash);transition:color .15s ease}
.nav:hover{background:rgba(255,255,255,.04);color:var(--bone)}
.nav:hover svg{color:var(--ink-2)}
.nav.on{background:var(--ember-12);border-color:rgba(242,84,27,.28);color:var(--bone);font-weight:600}
.nav.on svg{color:var(--ember)}
.nav.on::before{content:"";position:absolute;left:-13px;top:10px;bottom:10px;width:3px;border-radius:0 3px 3px 0;background:var(--ember)}
.nav.small{min-height:36px;font-size:var(--t-base);color:var(--ash)}
.nav.small svg{width:16px;height:16px;flex-basis:16px}
.nav.danger:hover{color:var(--ember)}
.nav.danger:hover svg{color:var(--ember)}
.nav.on.logs{color:var(--molten)}
.nav-dot{position:absolute;right:12px;top:50%;width:7px;height:7px;margin-top:-3px;border-radius:50%;background:var(--molten);box-shadow:0 0 8px rgba(255,179,92,.7)}
.rail-foot{margin-top:auto;display:flex;flex-direction:column;gap:2px;padding-top:12px;border-top:1px solid var(--line)}
.rail-status{font-size:var(--t-xs);color:var(--ash);padding:0 12px 10px;line-height:1.55;overflow:hidden;text-overflow:ellipsis;white-space:nowrap}
.rail-status b{color:var(--ink-2);font-weight:500}
.rail-version{font-size:var(--t-xs);color:var(--ash);padding:8px 12px 0;letter-spacing:.06em}
/* top bar */
.top{position:fixed;top:0;left:0;right:0;height:var(--top);display:flex;align-items:center;justify-content:space-between;gap:var(--s-4);padding:0 var(--gutter);background:rgba(12,12,14,.86);backdrop-filter:blur(12px);-webkit-backdrop-filter:blur(12px);border-bottom:1px solid rgba(42,42,48,.7);z-index:20;-webkit-app-region:drag}
.top button,.top .pill{-webkit-app-region:no-drag}
body.mac .top{padding-left:92px}
body.mac:not(.has-rail) .top{padding-left:92px}
body.has-rail .top{left:var(--rail)}
.brand{display:flex;align-items:center;gap:10px;min-width:0}
.brand .word{font-family:var(--head);font-weight:900;font-size:20px;letter-spacing:.06em}
.brand .miner{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.22em;color:var(--ash);margin-left:var(--s-1);padding-top:3px}
.top-right{display:flex;align-items:center;gap:var(--s-3);flex:0 0 auto}
.page-title{display:flex;align-items:baseline;gap:12px;min-width:0}
.page-title h1{font-size:19px;font-weight:700;letter-spacing:0;white-space:nowrap}
.page-sub{font-size:var(--t-base);color:var(--ash);white-space:nowrap;overflow:hidden;text-overflow:ellipsis;min-width:0}
.top-right{display:flex;align-items:center;gap:var(--s-3);flex:0 0 auto;min-width:0}
.top-status{font-size:var(--t-sm);color:var(--ash);white-space:nowrap;overflow:hidden;text-overflow:ellipsis;max-width:380px}
.top-status .pc+.pc::before{content:"·";margin:0 7px;color:var(--line-2)}
.pill{display:inline-flex;align-items:center;gap:var(--s-2);font-family:var(--mono);font-size:var(--t-sm);letter-spacing:.08em;text-transform:uppercase;color:var(--ash);border:1px solid var(--line);border-radius:999px;padding:6px 12px 6px 10px;background:var(--graphite);white-space:nowrap;font-variant-numeric:tabular-nums;min-width:112px;justify-content:center}
.pill.on{color:var(--molten);border-color:var(--molten-40)}
.pill.warn{color:var(--ember)}
@ -78,11 +108,12 @@ body.mac .top{padding-left:92px}
@keyframes pulse{0%,100%{box-shadow:0 0 0 0 rgba(255,179,92,.5)}50%{box-shadow:0 0 0 7px rgba(255,179,92,0)}}
@media (prefers-reduced-motion:reduce){.on .dot,.dot.live{animation:none}}
/* the notice strip under the top bar (app.js, Notices): one notice at a time. It takes no room while empty; main's
/* the status strip under the top bar (app.js, Notices): one notice at a time. It takes no room while empty; main's
top moves once per change with a 150 ms transition (layoutStrip), never per poll. Tones: default molten (running,
available, done), bad ember (failed, clock block, urgent). Nothing here is newer than 2022 CSS (WebView2 host). */
available, done), bad ember (failed, clock block, urgent). */
.notices{position:fixed;top:var(--top);left:0;right:0;z-index:19}
.notice{display:flex;flex-wrap:wrap;align-items:center;gap:var(--s-2) var(--s-4);padding:9px calc(var(--gutter) - 6px) 9px var(--gutter);background:var(--molten-10);border-bottom:1px solid var(--molten-40);font-size:var(--t-md);line-height:1.4;color:var(--bone)}
body.has-rail .notices{left:var(--rail)}
.notice{display:flex;flex-wrap:wrap;align-items:center;gap:var(--s-2) var(--s-4);padding:8px calc(var(--gutter) - 6px) 8px var(--gutter);background:var(--molten-10);border-bottom:1px solid var(--molten-40);font-size:var(--t-base);line-height:1.4;color:var(--bone)}
.notice.bad{background:var(--ember-12);border-bottom-color:var(--ember-40)}
.notice.warn{background:var(--molten-10);border-bottom-color:var(--molten-40)}
.notice.update-urgent{background:rgba(242,84,27,.55);border-bottom-color:var(--ember);color:#fff;font-weight:600}
@ -95,25 +126,18 @@ body.mac .top{padding-left:92px}
.notice-detail{flex-basis:100%;font-size:var(--t-sm);color:var(--ink-2);white-space:nowrap;overflow:hidden;text-overflow:ellipsis;margin-top:-3px}
.notice .prog{flex-basis:100%;height:3px;background:rgba(255,255,255,.12);border-radius:2px;overflow:hidden;margin-top:-3px}
.notice .prog i{display:block;height:100%;width:0;background:var(--ember);transition:width .5s linear}
.job-history table{margin-top:var(--s-2)}
.job-history th{text-align:left;font-weight:500;color:var(--ash);font-size:var(--t-xs);padding:4px 6px 4px 0}
.job-history td{padding:5px 6px 5px 0;border-top:1px solid var(--line);vertical-align:top}
.job-history tr.failed td,.job-history tr.timeout td,.job-history tr.aborted td{color:var(--ember)}
.job-history tr.running td{color:var(--molten)}
.clock-card{margin-top:var(--s-3);border:1px solid rgba(242,84,27,.5);background:rgba(242,84,27,.08);border-radius:12px;padding:12px 14px;display:flex;flex-direction:column;gap:var(--s-2)}
.clock-card.warn{border-color:var(--molten-40);background:rgba(255,179,92,.06)}
.clock-msg{font-size:var(--t-md);color:var(--bone);line-height:1.45}
.clock-card .note{margin-top:0}
/* screens */
/* screens and pages */
main{position:absolute;top:var(--top);bottom:0;left:0;right:0;overflow:auto;padding:0 var(--gutter);overscroll-behavior:contain;transition:top .15s ease}
@media (prefers-reduced-motion:reduce){main{transition:none}}
body.has-bottom main{bottom:var(--bottom)}
body.drawer-open main{bottom:calc(var(--bottom) + var(--drawer-h))}
body.has-rail main{left:var(--rail)}
body.drawer-open main{bottom:var(--drawer-h)}
.screen{display:none;max-width:1080px;margin:0 auto;animation:rise .45s ease}
body[data-phase="welcome"] #screen-welcome,body[data-phase="cards"] #screen-cards,body[data-phase="address"] #screen-address,body[data-phase="dashboard"] #screen-dashboard{display:block}
@keyframes rise{from{opacity:0;transform:translateY(12px)}to{opacity:1;transform:none}}
@media (prefers-reduced-motion:reduce){.screen{animation:none}}
@media (prefers-reduced-motion:reduce){.screen,.page{animation:none}}
#screen-dashboard{padding:22px 0 32px}
.page{display:flex;flex-direction:column;gap:var(--gap);animation:rise .3s ease}
/* welcome */
.hero{min-height:calc(100vh - var(--top));display:flex;flex-direction:column;align-items:center;justify-content:center;text-align:center;gap:var(--s-4);padding:var(--s-7) 0 48px}
@ -128,7 +152,6 @@ body[data-phase="welcome"] #screen-welcome,body[data-phase="cards"] #screen-card
.tile .k{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.14em;color:var(--ember)}
.tile .t{font-family:var(--head);font-weight:700;font-size:var(--t-lg);line-height:1.25}
.tile .s{font-size:var(--t-base);color:var(--ash);line-height:1.45}
.tile .h{font-family:var(--head);font-weight:700;font-size:var(--t-h3);margin-bottom:var(--s-2)}
.cta{display:flex;flex-wrap:wrap;gap:var(--s-3);align-items:center;justify-content:center;margin-top:10px}
.seedline{font-size:var(--t-sm);color:var(--ash);letter-spacing:.04em}
@ -138,6 +161,7 @@ body[data-phase="welcome"] #screen-welcome,body[data-phase="cards"] #screen-card
.step .cta{justify-content:flex-start;margin-top:var(--s-2)}
.note{font-size:var(--t-base);color:var(--ash);line-height:1.5}
.note.small{font-size:var(--t-sm);word-break:break-all}
.help{font-size:var(--t-base);color:var(--ash);line-height:1.5;max-width:64ch}
.cards{display:flex;flex-direction:column;gap:var(--s-3);margin-top:var(--s-2)}
.card{background:var(--graphite);border:1px solid var(--line);border-radius:var(--card-r);padding:var(--card-pad);min-width:0}
.card.detecting{display:flex;align-items:center;gap:18px}
@ -145,27 +169,14 @@ body[data-phase="welcome"] #screen-welcome,body[data-phase="cards"] #screen-card
.card.detecting .s{font-size:var(--t-sm);color:var(--ash);margin-top:var(--s-1)}
.spinner{width:28px;height:28px;border-radius:50%;border:3px solid var(--line-2);border-top-color:var(--ember);animation:spin 1s linear infinite;flex:0 0 28px}
@keyframes spin{to{transform:rotate(360deg)}}
.gpu{display:flex;align-items:center;gap:18px}
.gpu .badge{width:52px;height:52px;border-radius:14px;display:flex;align-items:center;justify-content:center;font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.08em;flex:0 0 52px;border:1px solid var(--line-2);color:var(--molten);background:var(--obsidian)}
.gpu .badge.apple{color:var(--bone)}
.gpu .badge.nvidia{color:var(--nvidia)}
.gpu .badge.amd{color:var(--ember)}
.gpu .name{font-family:var(--head);font-weight:700;font-size:var(--t-2xl);line-height:1.2}
.gpu .meta{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);margin-top:5px;display:flex;flex-wrap:wrap;gap:6px 14px}
.gpu .meta b{color:var(--ink-2);font-weight:500}
.gpu .tick{margin-left:auto;width:36px;height:36px;border-radius:50%;background:var(--ember);display:flex;align-items:center;justify-content:center;flex:0 0 36px}
.gpu.off .tick{background:var(--line-2)}
.gpu .tick svg path{stroke-dasharray:30;stroke-dashoffset:30;animation:draw .8s .2s ease forwards}
@keyframes draw{to{stroke-dashoffset:0}}
.gpu .msg{font-size:var(--t-sm);color:var(--ember);margin-top:var(--s-1)}
/* GPU rows (first run and settings) */
/* GPU rows (first run) */
.gpu-row{display:flex;align-items:center;gap:var(--s-4);background:var(--graphite);border:1px solid var(--line);border-radius:var(--card-r);padding:16px 20px;min-width:0;flex-wrap:wrap}
.gpu-row.off{opacity:.72}
.gpu-row .badge{width:48px;height:48px;border-radius:13px;display:flex;align-items:center;justify-content:center;font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.08em;flex:0 0 48px;border:1px solid var(--line-2);color:var(--molten);background:var(--obsidian)}
.gpu-row .badge.apple{color:var(--bone)}
.gpu-row .badge.nvidia{color:var(--nvidia)}
.gpu-row .badge.amd{color:var(--ember)}
.badge{width:48px;height:48px;border-radius:13px;display:flex;align-items:center;justify-content:center;font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.08em;flex:0 0 48px;border:1px solid var(--line-2);color:var(--molten);background:var(--obsidian)}
.badge.apple{color:var(--bone)}
.badge.nvidia{color:var(--nvidia)}
.badge.amd{color:var(--ember)}
.gpu-row .info{flex:1;min-width:180px;display:flex;flex-direction:column;gap:var(--s-1)}
.gpu-row .name{font-family:var(--head);font-weight:700;font-size:var(--t-xl);line-height:1.2;display:flex;align-items:center;gap:10px;flex-wrap:wrap}
.kind{font-family:var(--mono);font-size:10px;letter-spacing:.14em;text-transform:uppercase;border-radius:999px;padding:2px 8px;border:1px solid var(--line-2);color:var(--ash);font-weight:500;white-space:nowrap}
@ -175,56 +186,147 @@ body[data-phase="welcome"] #screen-welcome,body[data-phase="cards"] #screen-card
.gpu-row .meta b{color:var(--ink-2);font-weight:500}
.gpu-row .reason{font-size:var(--t-sm);color:var(--ash)}
.gpu-row .msg{font-size:var(--t-sm);color:var(--ember)}
.ids{display:flex;align-items:center;gap:6px;flex:0 0 auto}
.ids .k{font-family:var(--mono);font-size:10px;letter-spacing:.12em;text-transform:uppercase;color:var(--ash);margin-right:var(--s-1)}
.gpu-row .switch{padding:0;flex:0 0 auto;margin-left:auto}
/* hot-plug (src/hotplug.rs): a removed card dims, a faulty one is named in ember, neither has live controls */
.gpu-row.removed,.gpu-line.removed,.set-card.removed{opacity:.5}
.gpu-row.unusable .name,.gpu-line.unusable .name,.set-card.unusable .name{color:var(--ember)}
/* the Mine page: the big switch and the three numbers */
.hero-row{display:grid;grid-template-columns:1.3fr 1fr 1fr 1fr;gap:var(--gap-tile)}
.toggle-big{display:flex;flex-direction:column;align-items:flex-start;justify-content:center;gap:4px;min-height:112px;padding:18px 20px;border-radius:var(--tile-r);border:1px solid var(--ember);background:var(--ember);color:var(--ember-ink);cursor:pointer;text-align:left;transition:background .15s ease,border-color .15s ease,transform .15s ease,color .15s ease;min-width:0}
.toggle-big:hover{background:var(--ember-hi);border-color:var(--ember-hi);transform:translateY(-1px)}
.toggle-big:disabled{opacity:.45;cursor:default;transform:none}
.toggle-big .ring{width:34px;height:34px;border-radius:50%;display:flex;align-items:center;justify-content:center;background:rgba(12,12,14,.18);margin-bottom:6px}
.toggle-big .ring svg{width:18px;height:18px;fill:none;stroke:currentColor;stroke-width:2.2;stroke-linecap:round}
.toggle-big .tl{font-family:var(--head);font-weight:700;font-size:18px;line-height:1.15;white-space:nowrap}
.toggle-big .ts{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.06em;opacity:.8;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;max-width:100%}
.toggle-big.stop{background:var(--graphite);border-color:var(--line-2);color:var(--bone)}
.toggle-big.stop:hover{border-color:var(--ash);background:#1b1b20}
.toggle-big.stop .ring{background:var(--ember-12);color:var(--ember)}
.toggle-big.stop .ts{color:var(--molten);opacity:1}
.strip{display:grid;grid-template-columns:repeat(4,minmax(0,1fr));gap:var(--gap-tile)}
.strip.three{grid-template-columns:repeat(3,minmax(0,1fr))}
.cell{background:var(--graphite);border:1px solid var(--line);border-radius:var(--tile-r);padding:var(--tile-pad);display:flex;flex-direction:column;gap:6px;min-width:0}
.cell .k{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.12em;text-transform:uppercase;color:var(--ash);white-space:nowrap}
.cell .v{font-family:var(--head);font-weight:700;font-size:var(--t-num);line-height:1.1;font-variant-numeric:tabular-nums;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;display:flex;align-items:baseline;gap:var(--s-2);min-height:1.1em}
.cell.ember .v{color:var(--ember)}
.cell .v .unit{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);font-weight:500;letter-spacing:.08em}
.cell .v.state{text-transform:capitalize;font-size:22px}
.cell .v.small{font-size:18px}
.cell .v.dim{color:var(--ash)}
.cell .s{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);white-space:nowrap;overflow:hidden;text-overflow:ellipsis}
.cell.ok .v.state{color:var(--molten)}
.cell.bad .v.state{color:var(--ember)}
.cell.bad .s{color:var(--ember)}
.grid2{display:grid;grid-template-columns:1fr 1fr;gap:var(--gap);align-items:start}
.card-head{display:flex;justify-content:space-between;align-items:center;gap:var(--s-3);margin-bottom:var(--s-3);flex-wrap:wrap}
.stats{display:flex;flex-wrap:wrap;gap:12px 16px;font-size:var(--t-sm);color:var(--ash)}
.stats b{color:var(--bone);font-weight:500;font-variant-numeric:tabular-nums}
#dag{display:block;width:100%;height:170px;border-radius:12px;background:var(--obsidian)}
.legend{display:flex;flex-wrap:wrap;gap:14px 18px;margin-top:var(--s-3);font-size:var(--t-sm);color:var(--ink-2)}
.legend span{display:inline-flex;align-items:center;gap:var(--s-2)}
.sw{width:12px;height:12px;border-radius:3px;display:inline-block;border:1px solid var(--line-2)}
.sw.ember{background:var(--ember);border-color:var(--ember)}
.sw.glow{border-color:var(--molten);box-shadow:0 0 8px rgba(255,179,92,.7)}
.sw.line{width:18px;height:0;border:0;border-top:1px dashed var(--line-2);border-radius:0}
.empty{color:var(--ash);font-size:var(--t-base);padding:10px 0}
.empty.err{color:var(--ember)}
.kv{display:flex;flex-direction:column}
.kv>div{display:flex;justify-content:space-between;align-items:baseline;gap:var(--s-3);padding:7px 0;border-bottom:1px solid var(--line)}
.kv>div:last-child{border-bottom:0}
.kv .k{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.1em;text-transform:uppercase;color:var(--ash);white-space:nowrap}
.kv .v{font-size:var(--t-md);font-variant-numeric:tabular-nums;color:var(--bone);white-space:nowrap;overflow:hidden;text-overflow:ellipsis;text-align:right}
.kv .v.warn{color:var(--ember)}
.card .note{margin-top:10px}
.card .help+.help{margin-top:6px}
.big-word{font-family:var(--head);font-weight:700;font-size:20px;line-height:1.2;margin:2px 0 8px;word-break:break-word}
.big-word.ok{color:var(--molten)}
.big-word.warn{color:var(--ember)}
.big-word.dim{color:var(--ash)}
.feed{display:flex;flex-direction:column;font-family:var(--mono);font-size:var(--t-sm);color:var(--ink-2);max-height:260px;overflow:auto}
.feed>div{display:flex;justify-content:space-between;gap:10px;border-bottom:1px solid var(--line);padding:8px 0;align-items:baseline}
.feed>div:last-child{border-bottom:0}
.feed>div>span:first-child{min-width:0}
.feed .t{color:var(--ash);flex:0 0 auto;font-size:var(--t-xs)}
.feed .k{color:var(--molten);margin-right:6px}
.feed .k.error{color:var(--ember)}
.feed .k.warn{color:var(--ember)}
.feed .k.block{color:var(--ember)}
.feed .k.build{color:var(--ink-2)}
/* the GPU rows on the Mine page: badge, name, numbers, the switch */
.gpu-list{display:flex;flex-direction:column;gap:var(--s-2)}
.gpu-line{display:grid;grid-template-columns:44px minmax(160px,1.4fr) repeat(3,minmax(84px,.7fr)) 48px;align-items:center;gap:var(--s-4);padding:12px 14px;border:1px solid var(--line);border-radius:var(--tile-r);background:var(--obsidian);min-width:0}
.gpu-line.off{opacity:.62}
.gpu-line .badge{width:44px;height:44px;flex-basis:44px;border-radius:12px}
.gpu-line .who{min-width:0;display:flex;flex-direction:column;gap:3px}
.gpu-line .name{font-family:var(--head);font-weight:700;font-size:var(--t-md);line-height:1.25;display:flex;align-items:center;gap:8px;flex-wrap:wrap}
.gpu-line .st{display:inline-flex;align-items:center;gap:7px;font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);white-space:nowrap;overflow:hidden;text-overflow:ellipsis;min-width:0}
.gpu-line .st.on{color:var(--molten)}
.gpu-line .st.bad{color:var(--ember)}
.gpu-line .st i{width:7px;height:7px;border-radius:50%;background:currentColor;display:inline-block;flex:0 0 7px}
.gpu-line .msg{font-size:var(--t-sm);color:var(--ember);line-height:1.4}
.gpu-line .msg.dim{color:var(--ash)}
.gpu-line .num{display:flex;flex-direction:column;gap:2px;min-width:0}
.gpu-line .num .k{font-family:var(--mono);font-size:10px;letter-spacing:.12em;text-transform:uppercase;color:var(--ash);white-space:nowrap}
.gpu-line .num .v{font-family:var(--head);font-weight:700;font-size:18px;line-height:1.15;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;display:flex;align-items:baseline;gap:5px}
.gpu-line .num .v .unit{font-family:var(--mono);font-size:10px;color:var(--ash);font-weight:500;letter-spacing:.08em}
.gpu-line .num .v.hot{color:var(--ember)}
.gpu-line .num .v.warm{color:var(--molten)}
.gpu-line .num .v.na{color:var(--ash);font-weight:500;font-size:var(--t-md)}
.gpu-line.on .num.hash .v{color:var(--ember)}
.gpu-line .switch{padding:0;justify-self:end}
/* the Settings page: per-card controls */
.set-cards{display:flex;flex-direction:column;gap:var(--s-3);margin-bottom:var(--s-3)}
.set-card{border:1px solid var(--line);border-radius:var(--tile-r);background:var(--obsidian);padding:14px 16px;display:flex;flex-direction:column;gap:10px;min-width:0}
.set-card .head{display:flex;align-items:center;gap:10px;flex-wrap:wrap}
.set-card .head .name{font-family:var(--head);font-weight:700;font-size:var(--t-md)}
.set-card .head .meta{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash)}
.set-card .ctl{display:grid;grid-template-columns:150px 1fr auto;align-items:center;gap:var(--s-3)}
.set-card .ctl .k{font-size:var(--t-md);color:var(--bone);font-weight:500}
.set-card .ctl .pv{font-family:var(--mono);font-size:var(--t-sm);color:var(--ink-2);min-width:110px;text-align:right;white-space:nowrap}
.set-card input[type=range]{width:100%;accent-color:var(--ember);margin:0}
.set-card .ctl .help{grid-column:1 / -1;margin-top:-4px}
.set-card .line{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);display:flex;flex-wrap:wrap;gap:6px 12px;align-items:center}
.set-card .line b{color:var(--ink-2);font-weight:500}
.set-card .line.ok{color:var(--molten)}
.set-card .line.hot{color:var(--ember)}
.set-card .line.on{color:var(--molten)}
.set-card .line .pin{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.08em;text-transform:uppercase;color:var(--ash);border:1px solid var(--line-2);border-radius:6px;padding:1px 6px}
.ids{display:flex;align-items:center;gap:6px;flex:0 0 auto;justify-self:end}
.ids button{width:30px;height:30px;border-radius:8px;border:1px solid var(--line-2);background:transparent;color:var(--bone);cursor:pointer;font-size:16px;line-height:1}
.ids button:hover{border-color:var(--ash)}
.ids input{width:44px;text-align:center;background:var(--obsidian);border:1px solid var(--line-2);border-radius:8px;padding:5px 4px;color:var(--bone);font-family:var(--mono);font-size:var(--t-base);user-select:text;-webkit-user-select:text}
.ids input:focus{outline:none;border-color:var(--ember)}
.ids input::-webkit-inner-spin-button,.ids input::-webkit-outer-spin-button{-webkit-appearance:none;margin:0}
.gpu-row.off .ids{opacity:.4;pointer-events:none}
.gpu-row .switch{padding:0;flex:0 0 auto}
.cards.compact .gpu-row{padding:12px 14px;gap:var(--s-3);border-radius:14px}
.cards.compact .gpu-row .badge{width:38px;height:38px;flex-basis:38px;border-radius:10px}
.cards.compact .gpu-row .name{font-size:var(--t-md)}
.cards.compact .gpu-row .info{flex-basis:calc(100% - 50px);min-width:120px}
.cards.compact .gpu-row .power{margin-left:50px}
.cards.compact .gpu-row .switch{margin-left:auto}
.cards.compact .ids .k{display:none}
.power{display:flex;align-items:center;gap:var(--s-2);flex:0 0 auto}
.power .k{font-family:var(--mono);font-size:10px;letter-spacing:.12em;text-transform:uppercase;color:var(--ash)}
.power input[type=range]{width:110px;accent-color:var(--ember)}
.power .pv{font-size:var(--t-sm);color:var(--ink-2);min-width:84px}
.power .pin{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.08em;text-transform:uppercase;color:var(--ash);border:1px solid var(--line-2);border-radius:6px;padding:1px 6px}
.gpu-tile .msg.sweep{color:var(--ash)}
.gpu-tile .msg.sweep b{color:var(--ink-2);font-weight:500}
.gpu-tile .msg.sweep.on{color:var(--molten)}
.cards.compact .power .k{display:none}
.cards.compact .power input[type=range]{width:80px}
.gpu-tile .m.tele .warm,.gpu-tile .m.tele .warm b{color:var(--molten)}
.gpu-tile .m.tele .hot,.gpu-tile .m.tele .hot b{color:var(--ember)}
.gpu-tile .msg.ok,.gpu-row .msg.ok{color:var(--molten)}
.gpu-row .msg.hot,.gpu-tile .msg.hot{color:var(--ember)}
.gpu-tile .msg .btn.tiny,.gpu-row .msg .btn.tiny{margin-left:6px;vertical-align:middle}
.gpu-tile .msg.warm{color:var(--molten)}
.gpu-tile .msg.hot{color:var(--ember)}
/* per-card tiles on the dashboard */
.gpu-tiles{display:grid;grid-template-columns:repeat(auto-fit,minmax(220px,1fr));gap:var(--gap-tile)}
.gpu-tile{background:var(--obsidian);border:1px solid var(--line);border-radius:var(--tile-r);padding:14px 16px;display:flex;flex-direction:column;gap:6px;min-width:0}
.gpu-tile .n{font-family:var(--head);font-weight:700;font-size:var(--t-md);line-height:1.25;display:flex;align-items:center;gap:var(--s-2);flex-wrap:wrap}
.gpu-tile .h{font-family:var(--head);font-weight:700;font-size:24px;color:var(--ember);line-height:1.1;display:flex;align-items:baseline;gap:6px;font-variant-numeric:tabular-nums}
.gpu-tile .h .unit{font-family:var(--mono);font-size:var(--t-xs);color:var(--ash);font-weight:500;letter-spacing:.08em}
.gpu-tile.idle .h{color:var(--ash)}
.gpu-tile .m{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);display:flex;flex-wrap:wrap;gap:4px 12px}
.gpu-tile .m b{color:var(--ink-2);font-weight:500}
.gpu-tile .st{display:inline-flex;align-items:center;gap:7px;font-family:var(--mono);font-size:var(--t-sm);color:var(--ash)}
.gpu-tile .st.mining{color:var(--molten)}
.gpu-tile .st.bad{color:var(--ember)}
.gpu-tile .st i{width:7px;height:7px;border-radius:50%;background:currentColor;display:inline-block}
.gpu-tile .msg{font-size:var(--t-sm);color:var(--ash)}
.gpu-off{margin-top:var(--s-3);font-size:var(--t-sm);color:var(--ash)}
.lead-card .lead-row,.lead-row{display:flex;align-items:flex-start;justify-content:space-between;gap:var(--s-4)}
.lead-text{min-width:0;display:flex;flex-direction:column;gap:6px}
.lead-text h3{font-size:var(--t-2xl)}
.switch-list{display:flex;flex-direction:column;gap:3px;font-size:var(--t-sm);color:var(--ash);margin-top:8px}
.switch-list span{display:flex;justify-content:space-between;gap:12px}
.switch-list span.past{opacity:.55}
.switch-list span.next{color:var(--molten)}
.addr-big{display:flex;align-items:center;gap:12px;background:var(--obsidian);border:1px solid var(--line-2);border-radius:12px;padding:14px 16px;font-size:var(--t-lg);word-break:break-all;user-select:text;-webkit-user-select:text;margin-bottom:10px}
.addr-big span{flex:1;min-width:0;color:var(--molten)}
.card.adv{padding:0}
.card.adv summary{list-style:none;cursor:pointer;display:flex;align-items:center;justify-content:space-between;gap:12px;padding:var(--card-pad)}
.card.adv summary::-webkit-details-marker{display:none}
.card.adv summary::after{content:"+";font-family:var(--mono);color:var(--ash);font-size:18px;margin-left:auto}
.card.adv[open] summary::after{content:"\2212"}
.card.adv summary .eyebrow{margin-left:0}
.card.adv>:not(summary){margin-left:var(--card-pad);margin-right:var(--card-pad)}
.card.adv>:last-child{margin-bottom:var(--card-pad)}
.job-history table{margin-top:var(--s-2)}
table{border-collapse:collapse;width:100%;font-size:var(--t-base)}
.job-history th{text-align:left;font-weight:500;color:var(--ash);font-size:var(--t-xs);padding:4px 6px 4px 0;font-family:var(--mono);letter-spacing:.1em;text-transform:uppercase}
.job-history td{padding:6px 6px 6px 0;border-top:1px solid var(--line);vertical-align:top}
.job-history tr.failed td,.job-history tr.timeout td,.job-history tr.aborted td{color:var(--ember)}
.job-history tr.running td{color:var(--molten)}
.clock-card{border:1px solid rgba(242,84,27,.5);background:rgba(242,84,27,.08);border-radius:12px;padding:12px 14px;display:flex;flex-direction:column;gap:var(--s-2)}
.clock-card.warn{border-color:var(--molten-40);background:rgba(255,179,92,.06)}
.clock-msg{font-size:var(--t-md);color:var(--bone);line-height:1.45}
.clock-card .note{margin-top:0}
/* address options */
.options{display:flex;flex-direction:column;gap:var(--s-3);margin-top:6px}
@ -244,63 +346,8 @@ body[data-phase="welcome"] #screen-welcome,body[data-phase="cards"] #screen-card
.option:not(.on) .addr-input{display:none}
.err{font-size:var(--t-base);color:var(--ember)}
/* dashboard */
#screen-dashboard{padding:22px 0 28px;display:none;flex-direction:column;gap:var(--gap)}
body[data-phase="dashboard"] #screen-dashboard{display:flex}
.strip{display:grid;grid-template-columns:repeat(4,minmax(0,1fr));gap:var(--gap-tile)}
.cell{background:var(--graphite);border:1px solid var(--line);border-radius:var(--tile-r);padding:var(--tile-pad);display:flex;flex-direction:column;gap:6px;min-width:0}
.cell .k{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.12em;text-transform:uppercase;color:var(--ash);white-space:nowrap}
.cell .v{font-family:var(--head);font-weight:700;font-size:var(--t-num);line-height:1.1;font-variant-numeric:tabular-nums;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;display:flex;align-items:baseline;gap:var(--s-2);min-height:1.1em}
.cell.ember .v{color:var(--ember)}
.cell .v .unit{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);font-weight:500;letter-spacing:.08em}
.cell .v.state{text-transform:capitalize;font-size:22px}
.cell .s{font-family:var(--mono);font-size:var(--t-sm);color:var(--ash);white-space:nowrap;overflow:hidden;text-overflow:ellipsis}
.cell.ok .v.state{color:var(--molten)}
.cell.bad .v.state{color:var(--ember)}
.cell.bad .s{color:var(--ember)}
.grid{display:grid;grid-template-columns:1.35fr .85fr;gap:var(--gap);align-items:start}
.col-main,.col-side{display:flex;flex-direction:column;gap:var(--gap);min-width:0}
.card-head{display:flex;justify-content:space-between;align-items:center;gap:var(--s-3);margin-bottom:var(--s-3);flex-wrap:wrap}
.stats{display:flex;flex-wrap:wrap;gap:12px 16px;font-size:var(--t-sm);color:var(--ash)}
.stats b{color:var(--bone);font-weight:500;font-variant-numeric:tabular-nums}
#dag{display:block;width:100%;height:190px;border-radius:12px;background:var(--obsidian)}
.legend{display:flex;flex-wrap:wrap;gap:14px 18px;margin-top:var(--s-3);font-size:var(--t-sm);color:var(--ink-2)}
.legend span{display:inline-flex;align-items:center;gap:var(--s-2)}
.sw{width:12px;height:12px;border-radius:3px;display:inline-block;border:1px solid var(--line-2)}
.sw.ember{background:var(--ember);border-color:var(--ember)}
.sw.glow{border-color:var(--molten);box-shadow:0 0 8px rgba(255,179,92,.7)}
.sw.line{width:18px;height:0;border:0;border-top:1px dashed var(--line-2);border-radius:0}
.tbl{overflow-x:auto}
table{border-collapse:collapse;width:100%;font-size:var(--t-base)}
th,td{padding:9px 8px;text-align:left;border-bottom:1px solid var(--line);white-space:nowrap}
th{font-family:var(--mono);font-size:10px;letter-spacing:.12em;text-transform:uppercase;color:var(--ash);font-weight:500}
td.n,th.n{text-align:right;font-variant-numeric:tabular-nums;font-family:var(--mono)}
tr:last-child td{border-bottom:0}
td .st{display:inline-flex;align-items:center;gap:7px;font-family:var(--mono);font-size:var(--t-sm);color:var(--ash)}
td .st.mining{color:var(--molten)}
td .st.bad{color:var(--ember)}
td .st i{width:7px;height:7px;border-radius:50%;background:currentColor;display:inline-block}
td .sub{display:block;font-family:var(--mono);font-size:var(--t-xs);color:var(--ash);white-space:normal;max-width:280px}
.empty{color:var(--ash);font-size:var(--t-base);padding:10px 0}
.kv{display:flex;flex-direction:column}
.kv>div{display:flex;justify-content:space-between;align-items:baseline;gap:var(--s-3);padding:7px 0;border-bottom:1px solid var(--line)}
.kv>div:last-child{border-bottom:0}
.kv .k{font-family:var(--mono);font-size:var(--t-xs);letter-spacing:.1em;text-transform:uppercase;color:var(--ash);white-space:nowrap}
.kv .v{font-size:var(--t-md);font-variant-numeric:tabular-nums;color:var(--bone);white-space:nowrap;overflow:hidden;text-overflow:ellipsis;text-align:right}
.kv .v.warn{color:var(--ember)}
.card .note{margin-top:10px}
.feed{display:flex;flex-direction:column;font-family:var(--mono);font-size:var(--t-sm);color:var(--ink-2);max-height:300px;overflow:auto}
.feed>div{display:flex;justify-content:space-between;gap:10px;border-bottom:1px solid var(--line);padding:8px 0;align-items:baseline}
.feed>div:last-child{border-bottom:0}
.feed>div>span:first-child{min-width:0}
.feed .t{color:var(--ash);flex:0 0 auto;font-size:var(--t-xs)}
.feed .k{color:var(--molten);margin-right:6px}
.feed .k.error{color:var(--ember)}
.feed .k.block{color:var(--ember)}
.feed .k.build{color:var(--ink-2)}
/* the update card (app.js, UpdateCard; the wallet carries the same block): the mark with its ring, the name, one
line, up to three note lines, the size, Install now and Later. Over the key sheet and the settings panel. */
line, up to three note lines, the size, Install now and Later. Over the key sheet and the pages. */
.upd-wrap{position:fixed;inset:0;z-index:35;background:rgba(12,12,14,.72);backdrop-filter:blur(6px);-webkit-backdrop-filter:blur(6px);display:flex;align-items:center;justify-content:center;padding:24px;animation:fade .25s ease}
@keyframes fade{from{opacity:0}to{opacity:1}}
.upd-card{position:relative;width:100%;max-width:500px;max-height:calc(100vh - 48px);overflow:auto;background:var(--graphite);border:1px solid var(--line-2);border-radius:22px;padding:34px 32px 28px;display:flex;flex-direction:column;align-items:center;text-align:center;gap:var(--s-2);box-shadow:0 30px 80px rgba(0,0,0,.6),0 0 0 1px rgba(242,84,27,.08);animation:rise .3s ease;outline:none}
@ -334,7 +381,7 @@ td .sub{display:block;font-family:var(--mono);font-size:var(--t-xs);color:var(--
@media (max-height:620px){.upd-card{padding:24px 24px 20px}.upd-mark{width:72px;height:72px;margin-bottom:var(--s-2)}.upd-mark img{width:32px;height:32px}.upd-name{font-size:20px}}
/* sheet (the key) */
.sheet-wrap,.panel-wrap{position:fixed;inset:0;z-index:30;background:rgba(12,12,14,.72);backdrop-filter:blur(6px);-webkit-backdrop-filter:blur(6px);display:flex;align-items:center;justify-content:center;padding:24px}
.sheet-wrap{position:fixed;inset:0;z-index:30;background:rgba(12,12,14,.72);backdrop-filter:blur(6px);-webkit-backdrop-filter:blur(6px);display:flex;align-items:center;justify-content:center;padding:24px}
.sheet{background:var(--graphite);border:1px solid var(--line-2);border-radius:22px;padding:32px;max-width:640px;width:100%;max-height:calc(100vh - 48px);overflow:auto;display:flex;flex-direction:column;gap:14px;box-shadow:0 30px 80px rgba(0,0,0,.6);animation:rise .3s ease}
.sheet .sub{font-size:var(--t-lg);color:var(--ink-2)}
.field{display:flex;flex-direction:column;gap:var(--s-2);margin-top:6px}
@ -347,41 +394,29 @@ td .sub{display:block;font-family:var(--mono);font-size:var(--t-xs);color:var(--
.check input{width:18px;height:18px;accent-color:var(--ember);margin:0}
.check.small input{width:15px;height:15px}
.sheet .cta{justify-content:flex-start}
@media (prefers-reduced-motion:reduce){.sheet{animation:none}}
/* settings panel */
.panel-wrap{justify-content:flex-end;padding:0}
.panel{width:min(440px,100%);height:100%;background:var(--graphite);border-left:1px solid var(--line-2);display:flex;flex-direction:column;animation:slide .25s ease}
@keyframes slide{from{transform:translateX(30px);opacity:0}to{transform:none;opacity:1}}
@media (prefers-reduced-motion:reduce){.panel,.sheet{animation:none}}
.panel-head{display:flex;justify-content:space-between;align-items:center;padding:18px 22px;border-bottom:1px solid var(--line);flex:0 0 auto}
.panel-body{padding:14px 22px 28px;overflow:auto;display:flex;flex-direction:column;gap:18px;overscroll-behavior:contain}
/* rows, switches */
.row{display:flex;gap:10px;align-items:center;min-width:0}
.row.between{justify-content:space-between}
.row.wrap{flex-wrap:wrap}
.row .addr-input{margin-top:0;min-width:0}
.num{width:90px;background:var(--obsidian);border:1px solid var(--line-2);border-radius:10px;padding:9px 12px;color:var(--bone);font-size:var(--t-md);user-select:text;-webkit-user-select:text}
.num:focus{outline:none;border-color:var(--ember)}
.switch{display:flex;align-items:center;gap:var(--s-3);font-size:var(--t-md);cursor:pointer;padding:6px 0;line-height:1.35}
.switch{display:flex;align-items:center;gap:var(--s-3);font-size:var(--t-md);cursor:pointer;padding:8px 0 2px;line-height:1.35}
.switch input{position:absolute;opacity:0;width:0;height:0}
.switch .track{width:40px;height:22px;border-radius:999px;background:var(--line-2);position:relative;flex:0 0 40px;transition:background .15s ease}
.switch .track::after{content:"";position:absolute;top:3px;left:3px;width:16px;height:16px;border-radius:50%;background:var(--bone);transition:transform .15s ease}
.switch input:checked+.track{background:var(--ember)}
.switch input:checked+.track::after{transform:translateX(18px)}
.switch input:focus-visible+.track{outline:2px solid var(--ember);outline-offset:3px}
/* bottom bar */
.bottom{position:fixed;left:0;right:0;bottom:0;height:var(--bottom);display:flex;align-items:center;justify-content:space-between;gap:var(--s-3);padding:0 var(--gutter);background:rgba(12,12,14,.92);border-top:1px solid var(--line);z-index:21}
.bottom .left,.bottom .right{display:flex;align-items:center;gap:var(--s-2);flex:0 0 auto}
.bottom .mid{font-size:var(--t-sm);color:var(--ash);display:flex;align-items:center;justify-content:center;gap:0;min-width:0;flex:1;overflow:hidden;white-space:nowrap}
.bottom .mid .pc{flex:0 1 auto;min-width:0;overflow:hidden;text-overflow:ellipsis}
.bottom .mid .pc+.pc::before{content:"·";margin:0 7px;color:var(--line-2)}
.bottom .mid .nm{color:var(--ink-2);max-width:220px;flex-shrink:1}
.bottom .mid .ad{flex-shrink:0}
.bottom .right .mono{font-size:var(--t-sm)}
@media (max-width:1100px){.bottom .mid .up{display:none}.bottom .mid .nm{max-width:140px}.log-tools .at{display:none}.log-search{flex-basis:120px;max-width:200px}}
@media (max-width:960px){.bottom .mid .ad{display:none}}
.switch input:disabled+.track{opacity:.4}
.switch.lg .track{width:52px;height:30px;flex-basis:52px}
.switch.lg .track::after{width:24px;height:24px}
.switch.lg input:checked+.track::after{transform:translateX(22px)}
.switch+.help{margin-bottom:6px}
/* log drawer */
.drawer{position:fixed;left:0;right:0;bottom:var(--bottom);height:0;overflow:hidden;background:var(--obsidian);border-top:1px solid var(--line);z-index:20;transition:height .2s ease;display:flex;flex-direction:column;box-shadow:0 -12px 30px rgba(0,0,0,.35)}
.drawer{position:fixed;left:0;right:0;bottom:0;height:0;overflow:hidden;background:var(--obsidian);border-top:1px solid var(--line);z-index:21;transition:height .2s ease;display:flex;flex-direction:column;box-shadow:0 -12px 30px rgba(0,0,0,.35)}
body.has-rail .drawer{left:var(--rail)}
body.drawer-open .drawer{height:var(--drawer-h)}
body.drawer-drag .drawer{transition:none}
body.drawer-drag{cursor:row-resize}
@ -435,28 +470,56 @@ body.drawer-drag main{pointer-events:none}
.log-jump{position:absolute;right:calc(var(--gutter) + 14px);bottom:14px;background:var(--graphite);border-color:var(--molten-40);color:var(--molten);box-shadow:0 8px 24px rgba(0,0,0,.5);z-index:2;gap:6px;padding:6px 12px}
.log-jump:hover{background:#1c1a18;border-color:var(--molten)}
.toast{position:fixed;left:50%;bottom:calc(var(--bottom) + 16px);transform:translateX(-50%);background:var(--graphite);border:1px solid var(--line-2);border-radius:10px;padding:10px 16px;font-size:var(--t-base);z-index:40;box-shadow:0 10px 30px rgba(0,0,0,.5);max-width:min(90vw,520px);text-align:center}
body.drawer-open .toast{bottom:calc(var(--bottom) + var(--drawer-h) + 16px)}
.toast{position:fixed;left:50%;bottom:16px;transform:translateX(-50%);background:var(--graphite);border:1px solid var(--line-2);border-radius:10px;padding:10px 16px;font-size:var(--t-base);z-index:40;box-shadow:0 10px 30px rgba(0,0,0,.5);max-width:min(90vw,520px);text-align:center}
body.has-rail .toast{left:calc(50% + var(--rail) / 2)}
body.drawer-open .toast{bottom:calc(var(--drawer-h) + 16px)}
/* narrow windows: the strip folds to two columns, the side column drops under the main one */
@media (max-width:860px){
/* narrow windows (the 900 px floor): the rail folds to icons with small labels, the hero row and the strips fold to
two columns, the two-column grid to one, the GPU rows drop a number */
@media (max-width:1180px){
.hero-row{grid-template-columns:1fr 1fr}
.toggle-big{grid-row:span 1}
.strip{grid-template-columns:repeat(2,minmax(0,1fr))}
.grid{grid-template-columns:1fr}
.strip.three{grid-template-columns:repeat(3,minmax(0,1fr))}
.grid2{grid-template-columns:1fr}
.top-status{max-width:220px}
}
@media (max-width:1000px){
:root{--rail:76px;--gutter:var(--s-5)}
.rail{padding:12px 8px}
.rail-brand{justify-content:center;padding:6px 0 14px}
.rail-brand .word{display:none}
.nav{flex-direction:column;gap:4px;padding:8px 4px;font-size:10px;letter-spacing:.06em;text-transform:uppercase;font-family:var(--mono);font-weight:500;text-align:center;min-height:52px;justify-content:center}
.nav.on{font-weight:500}
.nav.on::before{left:-9px}
.nav.small{min-height:44px}
.nav-dot{right:8px;top:8px;margin:0}
.rail-status{display:none}
.rail-version{text-align:center;padding:8px 0 0;font-size:10px;white-space:nowrap}
.rail-version .chain{display:none}
.gpu-line{grid-template-columns:44px minmax(140px,1.4fr) repeat(2,minmax(80px,.7fr)) 48px}
.gpu-line .num.power{display:none}
.page-sub{display:none}
.top-status{display:none}
.strip.three{grid-template-columns:repeat(2,minmax(0,1fr))}
.set-card .ctl{grid-template-columns:120px 1fr auto}
}
@media (max-width:860px){
.three{grid-template-columns:1fr}
h1{font-size:40px}
}
/* short windows (under 700 px): tighter paddings, a lower hero, a smaller canvas, so the bottom bar and the drawer
always have room */
/* short windows (the 600 px floor): tighter paddings, a lower hero, a smaller canvas, so the drawer always has room */
@media (max-height:700px){
:root{--card-pad:18px;--tile-pad:14px 16px;--gap:var(--s-4)}
#screen-dashboard{padding:16px 0 20px}
#dag{height:150px}
#dag{height:140px}
.hero{gap:var(--s-3);padding:var(--s-5) 0 var(--s-6)}
.coin-wrap{width:104px;height:104px}
.coin{width:104px;height:104px}
h1{font-size:40px}
.lead{font-size:var(--t-xl)}
.step{padding:32px 0 32px}
.feed{max-height:220px}
.feed{max-height:200px}
.drawer-head{padding-bottom:6px}
.toggle-big{min-height:96px}
}

File diff suppressed because it is too large Load diff

View file

@ -8,24 +8,47 @@
<link rel="icon" href="mark.svg" type="image/svg+xml">
<link rel="stylesheet" href="app.css">
</head>
<body class="phase-welcome" data-phase="welcome">
<body class="phase-welcome" data-phase="welcome" data-page="mine">
<!-- the rail: one button per section (miner-ui-2). Shown on the dashboard; the setup screens have no rail. -->
<aside class="rail" id="rail" hidden>
<div class="rail-brand">
<img src="mark.svg" width="28" height="28" alt="">
<span class="word">IGNEUM</span>
</div>
<nav class="rail-nav" id="rail-nav" aria-label="Sections">
<button class="nav on" data-page="mine" aria-current="page"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M13 2 4 14h7l-1 8 9-12h-7l1-8z"/></svg><span>Mine</span></button>
<button class="nav" data-page="prove"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 2 4 5v6c0 5 3.4 9.4 8 11 4.6-1.6 8-6 8-11V5l-8-3z"/><path d="m9 12 2 2 4-4"/></svg><span>Prove</span></button>
<button class="nav" data-page="rewards"><svg viewBox="0 0 24 24" aria-hidden="true"><rect x="3" y="6" width="18" height="13" rx="2"/><path d="M3 10h18M16 15h2"/></svg><span>Rewards</span></button>
<button class="nav" data-page="node"><svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="12" cy="12" r="9"/><path d="M3 12h18M12 3c3 3.5 3 14.5 0 18M12 3c-3 3.5-3 14.5 0 18"/></svg><span>Node</span></button>
<button class="nav" data-page="updates"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 4v11M7 10l5 5 5-5"/><path d="M4 19h16"/></svg><span>Updates</span><i class="nav-dot" id="nav-updates-dot" hidden></i></button>
<button class="nav" data-page="settings"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M4 7h10M18 7h2M4 17h4M12 17h8"/><circle cx="16" cy="7" r="2"/><circle cx="10" cy="17" r="2"/></svg><span>Settings</span></button>
</nav>
<div class="rail-foot">
<div class="rail-status mono" id="rail-status"></div>
<button class="nav small" id="btn-logs" aria-expanded="false" aria-controls="drawer"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M4 6h16M4 12h16M4 18h10"/></svg><span id="btn-logs-text">Logs</span></button>
<button class="nav small danger" id="btn-quit"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 3v9"/><path d="M6.5 6.5a8 8 0 1 0 11 0"/></svg><span>Quit</span></button>
<div class="rail-version mono" id="foot-version"></div>
</div>
</aside>
<header class="top">
<div class="brand">
<div class="brand" id="top-brand">
<img src="mark.svg" width="30" height="30" alt="">
<span class="word">IGNEUM</span><span class="miner">MINER</span>
</div>
<div class="page-title" id="page-title" hidden>
<h1 id="page-title-text">Mine</h1>
<span class="page-sub" id="page-sub"></span>
</div>
<div class="top-right">
<span class="top-status mono" id="top-status"></span>
<div class="pill" id="pill"><span class="dot"></span><span id="pill-text">starting</span></div>
<button class="icon-btn" id="btn-settings" title="Settings" aria-label="Settings" hidden>
<svg viewBox="0 0 24 24" width="18" height="18" fill="none" stroke="currentColor" stroke-width="1.8" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><circle cx="12" cy="12" r="3"></circle><path d="M19.4 15a1.7 1.7 0 0 0 .3 1.8l.1.1a2 2 0 1 1-2.8 2.8l-.1-.1a1.7 1.7 0 0 0-1.8-.3 1.7 1.7 0 0 0-1 1.5V21a2 2 0 1 1-4 0v-.1a1.7 1.7 0 0 0-1.1-1.5 1.7 1.7 0 0 0-1.8.3l-.1.1a2 2 0 1 1-2.8-2.8l.1-.1a1.7 1.7 0 0 0 .3-1.8 1.7 1.7 0 0 0-1.5-1H3a2 2 0 1 1 0-4h.1a1.7 1.7 0 0 0 1.5-1.1 1.7 1.7 0 0 0-.3-1.8l-.1-.1a2 2 0 1 1 2.8-2.8l.1.1a1.7 1.7 0 0 0 1.8.3h.1a1.7 1.7 0 0 0 1-1.5V3a2 2 0 1 1 4 0v.1a1.7 1.7 0 0 0 1 1.5 1.7 1.7 0 0 0 1.8-.3l.1-.1a2 2 0 1 1 2.8 2.8l-.1.1a1.7 1.7 0 0 0-.3 1.8v.1a1.7 1.7 0 0 0 1.5 1H21a2 2 0 1 1 0 4h-.1a1.7 1.7 0 0 0-1.5 1z"></path></svg>
</button>
</div>
</header>
<!-- the notice strip: one notice at a time (updates, remote jobs, the clock), the most important first; app.js fills
it from the state and the content below moves once when it appears or goes. The close control hides a notice
until its state moves on. -->
<!-- the status strip: one notice at a time (updates, remote jobs, the clock), the most important first; app.js fills
it from the state (Notices) and the content below moves once when it appears or goes. -->
<div class="notices" id="notices" hidden>
<div class="notice" id="notice" role="status" aria-live="polite">
<span class="notice-text" id="notice-text"></span>
@ -85,8 +108,8 @@
</div>
</div>
<p class="note" id="cards-note" hidden></p>
<p class="note" id="cards-power" hidden>Igneum caps each NVIDIA card's power at 80% of its default limit to keep it stable (an RTX 5090 at full power hard-crashed in the field). This needs administrator rights once, when mining starts; the limit goes back to what it was on quit. The slider sets the cap per card. In the first hour, and then weekly, the efficiency sweep steps the cap from 100% down to 50% on the live program and holds the best MH per watt; a cap you set by hand stays pinned.</p>
<p class="note" id="cards-help" hidden>Each card you switch on gets its own worker. Identities take turns on the card and each one votes and pays separately; 8 suits a big card, 2 a small one, 1 an integrated GPU.</p>
<p class="note" id="cards-power" hidden>Igneum caps each NVIDIA card's power at 80% of its default limit to keep it stable. This needs administrator rights once, when mining starts; the limit goes back to what it was on quit. Settings has a slider per card.</p>
<p class="note" id="cards-help" hidden>Each card you switch on gets its own worker. An integrated GPU is off by default: it is slow and shares the machine's memory.</p>
<div class="cta">
<button class="btn primary" id="btn-cards-next" disabled>Continue</button>
<button class="btn ghost" id="btn-cards-retry" hidden>Detect again</button>
@ -128,64 +151,169 @@
</div>
</section>
<!-- 4. dashboard -->
<!-- 4. the dashboard: six pages behind the rail -->
<section class="screen" id="screen-dashboard">
<div class="strip">
<div class="cell big ember">
<div class="k">hash rate</div>
<div class="v"><span id="d-hash">0.0</span><span class="unit">MH/s</span></div>
<div class="s" id="d-hash-sub">waiting for the worker</div>
</div>
<div class="cell big">
<div class="k">blocks found</div>
<div class="v" id="d-blocks">0</div>
<div class="s" id="d-blocks-sub">accepted by the node</div>
</div>
<div class="cell big" id="d-node-cell">
<div class="k">node</div>
<div class="v state" id="d-node">starting</div>
<div class="s" id="d-node-sub">opening the database</div>
</div>
<div class="cell big">
<div class="k">next program</div>
<div class="v" id="d-eta">--:--</div>
<div class="s" id="d-eta-sub">waiting for the node</div>
</div>
</div>
<div class="grid">
<div class="col-main">
<div class="card">
<div class="card-head">
<div class="eyebrow"><span class="dot small"></span>your blocks</div>
<div class="stats mono"><span>last 10 min <b id="d-found-10">0</b></span><span>last hour <b id="d-found-60">0</b></span><span>dev fee <b id="d-fee">0</b></span><span>chain <b id="d-chain-blocks">0</b></span></div>
</div>
<canvas id="dag" aria-hidden="true"></canvas>
<div class="legend"><span><i class="sw ember"></i>block this machine found</span><span><i class="sw glow"></i>just accepted</span><span><i class="sw line"></i>one minute</span></div>
<!-- Mine -->
<section class="page" id="page-mine" data-page="mine">
<div class="hero-row">
<button class="toggle-big" id="btn-toggle" disabled>
<span class="ring"><svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 3v9"/><path d="M6.5 6.5a8 8 0 1 0 11 0"/></svg></span>
<span class="tl" id="btn-toggle-text">Start mining</span>
<span class="ts" id="btn-toggle-sub">waiting for the engine</span>
</button>
<div class="cell big ember">
<div class="k">hash rate</div>
<div class="v"><span id="d-hash">0.0</span><span class="unit">MH/s</span></div>
<div class="s" id="d-hash-sub">waiting for the worker</div>
</div>
<div class="card">
<div class="card-head"><h3>Cards</h3><div class="eyebrow" id="d-cards-eyebrow">1 worker</div></div>
<div class="gpu-tiles" id="d-cards"><div class="empty">No card yet.</div></div>
<div class="gpu-off mono" id="d-cards-off" hidden></div>
<div class="cell big">
<div class="k">blocks found</div>
<div class="v" id="d-blocks">0</div>
<div class="s" id="d-blocks-sub">accepted by the node</div>
</div>
<div class="cell big">
<div class="k">next program</div>
<div class="v" id="d-eta">--:--</div>
<div class="s" id="d-eta-sub">waiting for the node</div>
</div>
</div>
<div class="col-side">
<div class="card">
<div class="card-head"><h3>Your GPUs</h3><div class="eyebrow" id="d-cards-eyebrow">detecting</div></div>
<div class="gpu-list" id="d-cards"><div class="empty">Waiting for the engine.</div></div>
<p class="note" id="d-cards-note" hidden></p>
</div>
<div class="card">
<div class="card-head">
<h3>Your blocks</h3>
<div class="stats mono"><span>last 10 min <b id="d-found-10">0</b></span><span>last hour <b id="d-found-60">0</b></span><span>dev fee <b id="d-fee">0</b></span><span>chain <b id="d-chain-blocks">0</b></span></div>
</div>
<canvas id="dag" aria-hidden="true"></canvas>
<div class="legend"><span><i class="sw ember"></i>block this machine found</span><span><i class="sw glow"></i>just accepted</span><span><i class="sw line"></i>one minute</span></div>
</div>
<div class="card">
<div class="card-head"><h3>Activity</h3><div class="eyebrow">newest first</div></div>
<div class="feed" id="d-events"><div class="empty">No events yet.</div></div>
</div>
</section>
<!-- Prove -->
<section class="page" id="page-prove" data-page="prove" hidden>
<div class="card lead-card">
<div class="lead-row">
<div class="lead-text">
<h3>Prove shards on this machine</h3>
<p class="help">Every block on Igneum is turned into a short mathematical proof, in pieces called shards. The chain assigns shards to your keys; this machine proves them and is paid for each one. On by default on an NVIDIA card with 24 GB or more (a full shard needs 20.4 GB of GPU memory, measured); off on a Mac, whose CPU prover is slow.</p>
</div>
<label class="switch lg" title="Prove assigned shards"><input type="checkbox" id="s-prove" aria-label="Prove shards on this machine"><span class="track"></span></label>
</div>
<p class="note" id="pv-note">Off. Switch it on and this machine proves the shards the chain assigns to its keys.</p>
<div class="row" id="pv-setup-row" hidden><button class="btn small primary" id="pv-setup">Set up</button><span class="note">About 20 minutes, once.</span></div>
</div>
<div class="strip four">
<div class="cell"><div class="k">state</div><div class="v state" id="pv-state">off</div><div class="s" id="pv-state-sub">not proving</div></div>
<div class="cell"><div class="k">assigned</div><div class="v" id="pv-assigned">0</div><div class="s">shards given to your keys</div></div>
<div class="cell"><div class="k">proven</div><div class="v" id="pv-submitted">0</div><div class="s">proofs sent to the node</div></div>
<div class="cell"><div class="k">paid</div><div class="v" id="pv-paid">0</div><div class="s" id="pv-paid-sub">shards paid out</div></div>
</div>
<p class="note" id="pv-seg-note" hidden></p>
<div class="grid2">
<div class="card">
<div class="card-head"><h3>Node</h3><div class="eyebrow" id="d-node-net">devnet v4</div></div>
<div class="card-head"><h3>Verifier</h3><div class="eyebrow">the node's check</div></div>
<div class="big-word" id="pv-verifier">not read yet</div>
<p class="help" id="pv-verifier-help">Before a proof counts, the node checks it. A node without a verifier passes proofs along and never includes them.</p>
<p class="note" id="pv-verifier-note" hidden></p>
</div>
<div class="card">
<div class="card-head"><h3>Program</h3><div class="eyebrow">pinned guest</div></div>
<div class="field">
<div class="k">shard program id</div>
<div class="box mono"><span id="pv-program">not read yet</span><button class="btn tiny" data-copy="pv-program">Copy</button></div>
</div>
<div class="field">
<div class="k">aggregator id</div>
<div class="box mono"><span id="pv-aggregator">not read yet</span><button class="btn tiny" data-copy="pv-aggregator">Copy</button></div>
</div>
<p class="help">Every proof names the program that made it. Other nodes accept a proof only from these two ids.</p>
</div>
</div>
</section>
<!-- Rewards -->
<section class="page" id="page-rewards" data-page="rewards" hidden>
<div class="card">
<div class="card-head"><h3>Rewards address</h3><div class="eyebrow" id="r-source"></div></div>
<div class="addr-big mono"><span id="r-address">not set</span><button class="btn small" data-copy="r-address">Copy</button></div>
<p class="help" id="r-address-help">Every block this machine finds pays this address.</p>
</div>
<div class="strip three">
<div class="cell"><div class="k">blocks found</div><div class="v" id="r-blocks">0</div><div class="s">lifetime, accepted by the node</div></div>
<div class="cell"><div class="k">this run</div><div class="v" id="r-session">0</div><div class="s" id="r-session-sub">since the app started</div></div>
<div class="cell"><div class="k">balance</div><div class="v dim" id="r-balance">--</div><div class="s">shown in the wallet, not here yet</div></div>
</div>
<div class="grid2">
<div class="card" id="r-key-card">
<div class="card-head"><h3>Save your key</h3><div class="eyebrow ember">once</div></div>
<p class="help" id="r-key-help">The key for this address was made on this machine and is stored in the app folder, readable by your user only. Keep a copy somewhere safe: anyone with the key can spend what the address holds.</p>
<div class="row" id="r-key-row"><button class="btn small" id="s-reveal">Show my key</button><button class="btn small ghost" id="s-hide" hidden>Hide</button></div>
<div class="box mono key" id="s-key-box" hidden><span id="s-key"></span><button class="btn tiny" data-copy="s-key">Copy</button></div>
<p class="note mono small" id="r-key-file"></p>
</div>
<div class="card">
<div class="card-head"><h3>Use another address</h3></div>
<p class="help">Paste an EVM address you control. The miner restarts and pays the new address from the next block.</p>
<div class="row">
<input type="text" class="addr-input mono" id="s-address-input" placeholder="0x" spellcheck="false" autocomplete="off" aria-label="New rewards address">
<button class="btn small" id="s-address-save">Change</button>
</div>
<div class="err" id="s-address-err" hidden></div>
<p class="note" id="r-devfee-line"></p>
<div class="row"><button class="btn small ghost" id="r-wallet">Open the wallet page</button></div>
</div>
</div>
</section>
<!-- Node -->
<section class="page" id="page-node" data-page="node" hidden>
<div class="strip four">
<div class="cell" id="n-state-cell"><div class="k">node</div><div class="v state" id="d-node">starting</div><div class="s" id="d-node-sub">opening the database</div></div>
<div class="cell"><div class="k">height</div><div class="v" id="n-blocks">0</div><div class="s" id="n-blocks-sub">blocks this node holds</div></div>
<div class="cell"><div class="k">peers</div><div class="v" id="n-peers">0</div><div class="s" id="n-peers-sub">other nodes it talks to</div></div>
<div class="cell"><div class="k">version</div><div class="v small" id="n-version">--</div><div class="s" id="d-node-net">devnet v4</div></div>
</div>
<div class="clock-card" id="n-clock" hidden>
<p class="clock-msg" id="n-clock-msg"></p>
<div class="row"><button class="btn small primary" id="n-clock-sync">Sync clock</button><span class="note mono small" id="n-clock-result"></span></div>
<p class="note" id="n-clock-hint"></p>
</div>
<div class="grid2">
<div class="card">
<div class="card-head"><h3>Chain</h3><div class="eyebrow" id="n-reading">reading</div></div>
<div class="kv">
<div><span class="k">height</span><span class="v mono" id="n-blocks">0</span></div>
<div><span class="k">headers</span><span class="v mono" id="n-headers">0</span></div>
<div><span class="k">peers</span><span class="v mono" id="n-peers">0</span></div>
<div><span class="k">daa score</span><span class="v mono" id="n-daa">0</span></div>
<div><span class="k">difficulty</span><span class="v mono" id="n-diff">0</span></div>
<div><span class="k">tips</span><span class="v mono" id="n-tips">0</span></div>
<div><span class="k">blue score</span><span class="v mono" id="n-blue">0</span></div>
</div>
<div class="clock-card" id="n-clock" hidden>
<p class="clock-msg" id="n-clock-msg"></p>
<div class="row"><button class="btn small primary" id="n-clock-sync">Sync clock</button><span class="note mono small" id="n-clock-result"></span></div>
<p class="note" id="n-clock-hint"></p>
<p class="help">Headers arrive before blocks. The DAA score counts blocks the whole network made; the difficulty is how hard the next one is to find.</p>
</div>
<div class="card">
<div class="card-head"><h3>Consensus</h3><div class="eyebrow">the rules</div></div>
<div class="field">
<div class="k">digest</div>
<div class="box mono"><span id="n-digest">not printed yet</span><button class="btn tiny" data-copy="n-digest">Copy</button></div>
<p class="help">The fingerprint of the rules this node runs. Every node on the network shows the same one; a peer with another is refused.</p>
</div>
<div class="field">
<div class="k">next switch</div>
<div class="big-word" id="n-switch">none planned</div>
<p class="help" id="n-switch-help">A switch is a planned rule change. The node applies it by itself when the chain reaches that height.</p>
<div class="switch-list mono" id="n-switches"></div>
</div>
<p class="note" id="n-note"></p>
</div>
<div class="card">
<div class="card-head"><h3>Finality</h3><div class="eyebrow">miner-only</div></div>
@ -194,27 +322,95 @@
<div><span class="k">age</span><span class="v mono" id="f-age">n/a</span></div>
<div><span class="k">votes sent</span><span class="v mono" id="f-votes">0</span></div>
</div>
<p class="note" id="f-note">Locks appear once the miner votes on checkpoints.</p>
</div>
<div class="tile" id="tile-proving">
<div class="h">Proving</div>
<div class="kv">
<div><span class="k">state</span><span class="v mono" id="pv-state">off</span></div>
<div><span class="k">assigned</span><span class="v mono" id="pv-assigned">0</span></div>
<div><span class="k">submitted</span><span class="v mono" id="pv-submitted">0</span></div>
<div><span class="k">paid</span><span class="v mono" id="pv-paid">0</span></div>
<div><span class="k">verifier</span><span class="v mono" id="pv-verifier">not read yet</span></div>
</div>
<p class="note" id="pv-note">Off. Settings switches it on: this machine proves the shards the chain assigns to its keys.</p>
<p class="note" id="pv-verifier-note"></p>
<div class="row" id="pv-setup-row" hidden><button class="btn small" id="pv-setup">Set up</button></div>
<p class="help" id="f-note">A lock is a point the miners have agreed can never be undone. This machine votes on one every 30 s.</p>
</div>
<div class="card">
<div class="card-head"><h3>Events</h3><div class="eyebrow">newest first</div></div>
<div class="feed" id="d-events"><div class="empty">No events yet.</div></div>
<div class="card-head"><h3>Sync</h3><div class="eyebrow" id="n-sync-eyebrow"></div></div>
<div class="big-word" id="n-sync-word">starting</div>
<p class="help" id="n-note"></p>
</div>
</div>
</div>
</section>
<!-- Updates -->
<section class="page" id="page-updates" data-page="updates" hidden>
<div class="card lead-card">
<div class="lead-row">
<div class="lead-text">
<h3 id="s-version">Igneum Miner</h3>
<p class="help" id="s-update-note">Not checked yet.</p>
</div>
<div class="row">
<button class="btn small primary" id="s-install" hidden>Install now</button>
<button class="btn small" id="s-update">Check now</button>
</div>
</div>
<label class="switch"><input type="checkbox" id="s-auto-update"><span class="track"></span><span>Install updates by itself</span></label>
<p class="help">Downloads in the background and installs at a quiet moment, never mid-program. Off: it downloads, then waits for Install now.</p>
</div>
<div class="card">
<div class="lead-row">
<div class="lead-text">
<h3>Remote jobs</h3>
<p class="help">Igneum publishes signed jobs (a benchmark, a script, logs to collect) next to the update manifest. This machine runs each one once and reports back. Only jobs signed by Igneum's key run.</p>
</div>
<button class="btn small" id="s-jobs-check">Check now</button>
</div>
<p class="note" id="s-jobs-note"></p>
<div class="job-history" id="s-jobs-history"></div>
<p class="note mono small" id="s-jobs-key"></p>
</div>
</section>
<!-- Settings -->
<section class="page" id="page-settings" data-page="settings" hidden>
<div class="card">
<div class="card-head"><h3>Graphics cards</h3><div class="eyebrow" id="s-cards-eyebrow"></div></div>
<div class="set-cards" id="s-cards"><div class="empty">No card yet.</div></div>
<label class="switch"><input type="checkbox" id="s-sweep"><span class="track"></span><span>Ember Tune: tune every card for hashes per watt</span></label>
<p class="help">Once after install, then weekly and after a driver or program change: the power limit steps from 100% down to 50%, then the core clock from its maximum down to 60%, 75 s a step on the live program; the memory clock is never touched. The card keeps the point with the most hashes per watt within 1% of its top rate. A step with a rejected hash, a hot GPU or a dragged memory clock is reverted. A card whose model the fleet already knows starts at that point and confirms it in two steps. Every result goes back to the fleet without anything that identifies you. A cap you set by hand is left alone.</p>
<label class="switch"><input type="checkbox" id="s-power-control"><span class="track"></span><span>Power control: let the app set NVIDIA limits</span></label>
<p class="help">Windows asks for administrator rights once; the NVIDIA cap and the tune need them. Off, the app never asks and NVIDIA cards measure only. AMD cards need no rights. <span id="s-power-note"></span></p>
</div>
<div class="card">
<div class="card-head"><h3>This machine</h3></div>
<label class="switch"><input type="checkbox" id="s-login"><span class="track"></span><span>Start at login</span></label>
<p class="help">The miner opens when you sign in and keeps mining in the background.</p>
<label class="switch"><input type="checkbox" id="s-jobs-allow"><span class="track"></span><span>Allow remote jobs from Igneum</span></label>
<p class="help">Signed jobs from Igneum run on this machine and report back. Updates shows what ran.</p>
<label class="switch"><input type="checkbox" id="s-vote"><span class="track"></span><span>Vote on finality checkpoints</span></label>
<p class="help">Your miner signs a checkpoint every 30 s. Votes are what lock the chain; leave it on.</p>
<div class="field">
<div class="k">name</div>
<div class="row">
<input type="text" class="addr-input" id="s-name" placeholder="a name for this machine" maxlength="40" spellcheck="false" aria-label="Machine name">
<button class="btn small" id="s-name-save">Rename</button>
</div>
<p class="help">A label for you only. Keys come from the machine id <span class="mono" id="s-mid"></span>, never from the name.</p>
</div>
</div>
<div class="card">
<div class="card-head"><h3>Dev fee</h3></div>
<label class="switch"><input type="checkbox" id="s-devfee"><span class="track"></span><span id="s-devfee-text">Dev fee 1% (1 block in 100)</span></label>
<p class="help" id="s-devfee-note">One block in 100 is mined for the miner software's author, the same way every GPU miner takes a fee. The protocol itself takes nothing. This switch turns it off.</p>
</div>
<div class="card">
<div class="card-head"><h3>Logs</h3></div>
<div class="row wrap">
<button class="btn small" id="s-log-copy">Copy the log</button>
<button class="btn small ghost" id="s-log-open">Show the log</button>
</div>
<p class="help">Copies the last lines the node and the miner wrote, for a support message. Show the log opens the drawer with every line.</p>
<p class="note mono small" id="s-log-dir"></p>
<p class="note mono small" id="s-node-dir"></p>
</div>
<details class="card adv" id="s-advanced">
<summary><h3>Advanced</h3><span class="eyebrow">devnet tools</span></summary>
<label class="switch"><input type="checkbox" id="s-trust"><span class="track"></span><span>Trust proof records without verifying them</span></label>
<p class="help" id="s-trust-note">Devnet only. When no verifier is found next to the engine, the node includes proof records it never checked. A found verifier always wins. Changing this restarts the node.</p>
<div class="row"><button class="btn small ghost" id="s-live" hidden>Open the live devnet page</button></div>
</details>
</section>
</section>
</main>
@ -252,7 +448,7 @@
<div class="k">private key</div>
<div class="box mono key"><span id="key-private"></span><button class="btn tiny" data-copy="key-private">Copy</button></div>
</div>
<p class="note">Stored at <span class="mono" id="key-file"></span>, readable by your user only. Settings can show it again.</p>
<p class="note">Stored at <span class="mono" id="key-file"></span>, readable by your user only. Rewards can show it again.</p>
<p class="note">On devnet the vote keys are test keys derived from the miner's label. Mainnet vote keys will be random and stored like this wallet.</p>
<label class="check"><input type="checkbox" id="key-ack"><span>I have saved my key</span></label>
<div class="cta">
@ -261,75 +457,6 @@
</div>
</div>
<!-- settings -->
<div class="panel-wrap" id="settings" hidden>
<aside class="panel">
<div class="panel-head"><h3>Settings</h3><button class="icon-btn" id="btn-settings-close" aria-label="Close">&times;</button></div>
<div class="panel-body">
<div class="field">
<div class="k">rewards address</div>
<div class="box mono"><span id="s-address"></span><button class="btn tiny" data-copy="s-address">Copy</button></div>
<div class="row">
<input type="text" class="addr-input mono" id="s-address-input" placeholder="paste a new address" spellcheck="false" autocomplete="off">
<button class="btn small" id="s-address-save">Change</button>
</div>
<div class="err" id="s-address-err" hidden></div>
<button class="btn small ghost" id="s-reveal" hidden>Show my key</button>
<div class="box mono key" id="s-key-box" hidden><span id="s-key"></span><button class="btn tiny" data-copy="s-key">Copy</button></div>
<label class="switch"><input type="checkbox" id="s-devfee"><span class="track"></span><span id="s-devfee-text">Dev fee 1% (1 block in 100)</span></label>
<p class="note" id="s-devfee-note">The miner software's fee, the same way every GPU miner takes one. The protocol takes none. This switch turns it off.</p>
</div>
<div class="field">
<div class="k">this machine</div>
<div class="row">
<input type="text" class="addr-input" id="s-name" placeholder="a name for this machine" maxlength="40" spellcheck="false">
<button class="btn small" id="s-name-save">Rename</button>
</div>
<p class="note">A label for you only. Keys and labels come from the machine id <span class="mono" id="s-mid"></span>, never from the computer name.</p>
</div>
<div class="field">
<div class="k">cards</div>
<div class="cards compact" id="s-cards"></div>
<div class="row between"><p class="note">Switch a card on or off, set its identities. Only that card's worker restarts; the node keeps running.</p><button class="btn small" id="s-cards-save">Apply</button></div>
</div>
<div class="field">
<label class="switch"><input type="checkbox" id="s-vote"><span class="track"></span><span>Vote on finality checkpoints</span></label>
<label class="switch"><input type="checkbox" id="s-prove"><span class="track"></span><span>Prove assigned shards (proving v0; on a Mac the CPU prover is slow)</span></label>
<label class="switch"><input type="checkbox" id="s-trust"><span class="track"></span><span>Trust proof records without verifying them (devnet only)</span></label>
<p class="note" id="s-trust-note">Only when no verifier is found next to the engine: the node then includes proof records it never checked. Never on a testnet. A found verifier always wins. Changing this restarts the node.</p>
<label class="switch"><input type="checkbox" id="s-login"><span class="track"></span><span>Start at login</span></label>
</div>
<div class="field">
<div class="k">efficiency sweep</div>
<label class="switch"><input type="checkbox" id="s-sweep"><span class="track"></span><span>Find each NVIDIA card's best MH per watt (once after install, then weekly)</span></label>
<p class="note">On the live program, never restarting the worker: the cap steps from 100% of the card's default limit down to 50%, 15 s to settle and 60 s to measure per step, then holds the step with the most MH per watt. One administrator prompt per sweep. It stops at once if the card faults, a remote job takes the GPU, or the hour boundary is near. A cap you set with the slider is pinned: the sweep records, but leaves it. "Sweep now" on a card's tile runs one at any time; the table is in the log (SWEEP lines).</p>
</div>
<div class="field">
<div class="k">version</div>
<div class="row between"><span class="mono" id="s-version"></span><span class="row"><button class="btn small primary" id="s-install" hidden>Install now</button><button class="btn small" id="s-update">Check now</button></span></div>
<label class="switch"><input type="checkbox" id="s-auto-update"><span class="track"></span><span>Install updates by itself at a safe moment</span></label>
<p class="note" id="s-update-note"></p>
</div>
<div class="field">
<div class="k">remote jobs</div>
<div class="row between"><label class="switch"><input type="checkbox" id="s-jobs-allow"><span class="track"></span><span>Allow remote jobs from Igneum (signed)</span></label><button class="btn small" id="s-jobs-check">Check now</button></div>
<p class="note">Igneum publishes signed jobs (a benchmark, a script, a file to fetch, logs to collect, a restart) next to the update manifest. This machine runs each one once and reports to the Igneum log intake. Only jobs signed by the key below run; nothing else can send one.</p>
<p class="note mono small" id="s-jobs-key"></p>
<p class="note" id="s-jobs-note"></p>
<div class="job-history" id="s-jobs-history"></div>
</div>
<div class="field">
<div class="k">folders</div>
<p class="note mono small" id="s-node-dir"></p>
<p class="note mono small" id="s-log-dir"></p>
</div>
<div class="field">
<button class="btn small ghost" id="s-live" hidden>Open the live devnet page</button>
</div>
</div>
</aside>
</div>
<!-- the log drawer: chips filter by source, search filters by text, the ruler and the jump controls move in time.
The list is virtualised (fixed row height, only the visible rows are in the DOM) so 20,000 lines scroll smoothly. -->
<div class="drawer" id="drawer" aria-label="Logs">
@ -357,6 +484,7 @@
<button class="chip" id="log-wrap" title="Wrap long lines (up to 5,000 lines)">wrap</button>
<button class="chip" id="log-copy" title="Copy the lines in view">copy</button>
<label class="check small"><input type="checkbox" id="log-follow" checked><span>follow</span></label>
<button class="chip" id="log-close" title="Close the log">close</button>
</div>
</div>
<div class="log-ruler" id="log-ruler" title="Click to jump"><span class="t0 mono" id="log-ruler-0"></span><span class="t1 mono" id="log-ruler-1"></span><i class="win" id="log-ruler-win"></i></div>
@ -367,18 +495,6 @@
<button class="btn small log-jump" id="log-jump" hidden><svg viewBox="0 0 24 24" width="14" height="14" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><path d="M12 5v14M5 12l7 7 7-7"></path></svg><span id="log-jump-text">Newest</span></button>
</div>
<footer class="bottom" id="bottom" hidden>
<div class="left">
<button class="btn small" id="btn-pause">Pause</button>
<button class="btn small ghost" id="btn-logs" aria-expanded="false" aria-controls="drawer"><span id="btn-logs-text">Logs</span><svg class="chev" viewBox="0 0 24 24" width="14" height="14" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><path d="m6 15 6-6 6 6"></path></svg></button>
</div>
<div class="mid mono" id="foot-status" aria-live="off"></div>
<div class="right">
<span class="mono dim" id="foot-version"></span>
<button class="btn small ghost danger" id="btn-quit">Quit</button>
</div>
</footer>
<div class="toast" id="toast" hidden></div>
<script src="app.js"></script>
</body>

View file

@ -11,7 +11,7 @@ const src = readFileSync(join(dirname(fileURLToPath(import.meta.url)), 'app.js')
const mod = { exports: {} };
new Function('module', src)(mod);
const N = mod.exports;
const { updateNotice, jobNotice, clockNotice, gather, pick } = N;
const { updateNotice, jobNotice, clockNotice, cardNotices, gather, pick } = N;
const NOW = 1_800_000_000;
const upd = (over) => ({ available: true, version: '0.3.6', notes: '', checked_at: NOW - 60, error: '', status: 'ready', downloaded: true, ready: true, applying: false, progress: 1, size: 20_588_331, auto: true, wait: 'installs at the next safe moment', urgent: false, urgent_text: '', activation_height: 0, unsupported: false, min_supported: '', channel: 'devnet', published_at: '', file: '', updated_from: '', rolled_back: '', ...over });
@ -139,3 +139,36 @@ test('clock: the engine words, Sync clock, the hint; gather() keeps it off the d
assert.equal(pick(gather(st, { dashboard: false }), {}).kind, 'clock');
assert.deepEqual(gather({}, { dashboard: true }), []);
});
// hot-plug (src/hotplug.rs): the strip says what appeared or went, for 5 minutes, on every screen
const cardOf = (over) => ({ index: 0, key: 'nvidia:0:NVIDIA GeForce RTX 5090', kind: 'discrete', name: 'NVIDIA GeForce RTX 5090', vendor: 'nvidia', enabled: true, state: 'mining', problem: '', reason: '', message: '', added_at: 0, removed_at: 0, gone: false, ...over });
test('card notices: new card mining, new card not usable with the hint, removed card, nothing for the cards found at start', () => {
const start = cardOf();
assert.deepEqual(cardNotices([start], NOW), []);
const added = cardOf({ key: 'amd:1:gfx1201', name: 'gfx1201', vendor: 'amd', added_at: NOW - 30, state: 'starting' });
const [a] = cardNotices([start, added], NOW);
assert.equal(a.kind, 'card-added'); assert.equal(a.level, N.LEVEL['job-done']); assert.equal(a.text, 'New card: gfx1201, mining.'); assert.equal(a.key, 'card:added:amd:1:gfx1201:' + (NOW - 30));
const bad = cardOf({ key: 'amd::AMD Radeon RX 9070 XT', name: 'AMD Radeon RX 9070 XT', vendor: 'amd', enabled: false, state: 'unusable', problem: 'Code 43', message: 'not usable (Code 43)', reason: 'reboot with the card attached; if it persists, reinstall the driver with the card attached', added_at: NOW - 10 });
const [b] = cardNotices([bad], NOW);
assert.equal(b.text, 'AMD Radeon RX 9070 XT: not usable (Code 43). No worker runs on it.'); assert.equal(b.tone, 'warn'); assert.match(b.detail, /^reboot with the card attached/);
const igpu = cardOf({ key: 'amd:0:gfx1036', name: 'gfx1036', vendor: 'amd', kind: 'integrated', enabled: false, state: 'off', added_at: NOW - 5 });
assert.equal(cardNotices([igpu], NOW)[0].text, 'New card: gfx1036, off (integrated). Settings switches it on.');
const removed = cardOf({ key: 'amd:1:gfx1201', name: 'gfx1201', vendor: 'amd', state: 'removed', removed_at: NOW - 60 });
const [r] = cardNotices([removed], NOW);
assert.equal(r.kind, 'card-removed'); assert.equal(r.text, 'Card removed: gfx1201. Its worker stopped.'); assert.equal(r.tone, 'warn');
// the newest first; both go after CARD_S
const two = cardNotices([added, removed], NOW);
assert.equal(two[0].kind, 'card-added');
assert.deepEqual(cardNotices([added, removed, bad], NOW + N.CARD_S + 1), []);
});
test('card notices sit under a running job and show on the setup screens too', () => {
const added = cardOf({ key: 'amd:1:gfx1201', name: 'gfx1201', vendor: 'amd', added_at: NOW - 30 });
const withJob = s({ jobs: run(), mining: { cards: [added] } });
assert.equal(pick(gather(withJob, { dashboard: true }), {}).kind, 'job-running');
const quiet = s({ mining: { cards: [added] } });
assert.equal(pick(gather(quiet, { dashboard: true }), {}).kind, 'card-added');
assert.equal(pick(gather(quiet, {}), {}).kind, 'card-added');
// closed: stays closed for that key; a later event on the same card has a new key
assert.equal(pick(gather(quiet, { dashboard: true }), { ['card:added:amd:1:gfx1201:' + (NOW - 30)]: true }), null);
});

View file

@ -0,0 +1,44 @@
// node --test app/igneum-app/ui/tune-line.test.mjs (no dependencies; CI runs it in the site job)
// The card row's Ember Tune line (app.js TuneLine): what a user sees per state, from the card state fields.
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { readFileSync } from 'node:fs';
import { fileURLToPath } from 'node:url';
import { dirname, join } from 'node:path';
const src = readFileSync(join(dirname(fileURLToPath(import.meta.url)), 'app.js'), 'utf8');
const mod = { exports: {} };
new Function('module', src)(mod);
const { model, point } = mod.exports.TuneLine;
const NOW = 1_800_000_000;
const card = over => ({ vendor: 'nvidia', sweep_state: 'idle', sweep_note: '', sweep_pct: 0, power_pct: 80, sweep_at: 0, tune_line: '', tune_source: '', tune_clock_mhz: 0, tune_control: true, pinned: false, ...over });
test('tuned: the line the brief asks for, with the point, the source and when', () => {
const m = model(card({ tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'full', tune_clock_mhz: 2470, sweep_pct: 100, sweep_at: NOW - 3600 }), NOW);
assert.equal(m.kind, 'tuned');
assert.equal(m.text, 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)');
assert.equal(m.note, '2470 MHz at 100%, full tune, 1 h ago');
const c = model(card({ tune_line: 'Tuned: 17.7 MH/s at 177 W (0.100 MH/W)', tune_source: 'confirm', sweep_pct: 90, sweep_at: NOW - 120, vendor: 'amd' }), NOW);
assert.equal(c.note, '90%, clock unlocked, from the fleet prior, confirmed, 2 min ago');
const p = model(card({ tune_line: 'Tuned: 1 MH/s at 1 W (1.000 MH/W)', tune_source: 'full', sweep_pct: 70, pinned: true }), NOW);
assert.match(p.note, /your setting stays pinned$/);
});
test('measure only: Apple and NVIDIA without Power control say so beside the measured line', () => {
const a = model(card({ vendor: 'apple', tune_control: false, tune_line: 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W)', tune_source: 'baseline', sweep_note: 'measure only on Apple silicon: the system sets the clocks and the power; no control exposed', sweep_at: NOW - 60 }), NOW);
assert.equal(a.kind, 'measured');
assert.equal(a.text, 'Tuned: 26.7 MH/s at 38 W (0.703 MH/W) (measured as it runs, 1 min ago)');
assert.match(a.note, /^measure only on Apple silicon/);
const n = model(card({ tune_control: false, tune_line: 'Tuned: 122.3 MH/s at 290 W (0.422 MH/W)', tune_source: 'baseline', sweep_note: 'measure only until Power control is on in Settings (Windows asks for administrator rights once)' }), NOW);
assert.equal(n.kind, 'measured');
assert.match(n.note, /Power control/);
});
test('running, stopped, idle and off', () => {
assert.deepEqual(model(card({ sweep_state: 'running', sweep_note: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' }), NOW), { kind: 'running', text: 'tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)' });
assert.equal(model(card({ sweep_note: 'tuning stopped: a remote job took the GPU' }), NOW).kind, 'stopped');
assert.equal(model(card({}), NOW).text, 'tuning: not run yet (starts after 120 s of steady mining)');
assert.equal(model(card({ sweep_note: 'tuning: waits for 120 s of steady mining' }), NOW).text, 'tuning: waits for 120 s of steady mining');
assert.equal(model(card({ vendor: 'other' }), NOW).kind, 'off');
assert.equal(point({ tune_clock_mhz: 0, sweep_pct: 0, power_pct: 0 }), '');
});

View file

@ -0,0 +1,200 @@
// node --test app/igneum-app/ui/view.test.mjs (no dependencies; CI runs it in the site job)
// Loads the View block of app.js (plain browser JS: the file is run with `module` defined and no `document`, so only
// the pure blocks execute) and checks the words each page shows for a given state (miner-ui-2).
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { readFileSync } from 'node:fs';
import { fileURLToPath } from 'node:url';
import { dirname, join } from 'node:path';
const src = readFileSync(join(dirname(fileURLToPath(import.meta.url)), 'app.js'), 'utf8');
const mod = { exports: {} };
new Function('module', src)(mod);
const V = mod.exports.View;
const card = (over) => ({ key: 'nvidia:0:RTX 5090', name: 'NVIDIA GeForce RTX 5090', vendor: 'nvidia', kind: 'discrete', worker: 'CUDA', vram_mb: 32768, enabled: true, state: 'mining', hash_now: 124.3, hash_avg: 120, accepted: 3, rejected: 0, identities: 8, ids: [], prepared: true, restart_in_s: 0, message: '', reason: '', power_w: 410.2, power_limit_w: 460, power_default_w: 575, power_pct: 80, power_applied: true, temp_gpu: 61, temp_mem: 72, telemetry_at: 1, ...over });
test('the six sections and their order', () => {
assert.deepEqual(V.PAGES.map((p) => p.id), ['mine', 'prove', 'rewards', 'node', 'updates', 'settings']);
assert.equal(V.page('node').title, 'Node');
assert.equal(V.page('nonsense').id, 'mine');
});
test('a GPU row: name, kind, the numbers where known, the switch', () => {
const r = V.cardRow(card());
assert.equal(r.name, 'NVIDIA GeForce RTX 5090');
assert.equal(r.kindWord, 'Discrete');
assert.equal(r.integrated, false);
assert.equal(r.on, true);
assert.equal(r.word, 'mining');
assert.equal(r.tone, 'on');
assert.equal(r.hash, '124');
assert.equal(r.temp, '61 °C');
assert.equal(r.tempTone, '');
assert.equal(r.power, '410 W');
assert.equal(r.cap, 460);
assert.equal(r.meta, '32 GB · 3 blocks · next program ready');
assert.equal(r.canToggle, true);
});
test('a GPU row: temperatures turn amber then red; unknown numbers are empty', () => {
assert.equal(V.cardRow(card({ temp_mem: 92 })).tempTone, 'warm');
assert.equal(V.cardRow(card({ temp_mem: 96 })).tempTone, 'hot');
assert.equal(V.cardRow(card({ temp_gpu: 93 })).tempTone, 'hot');
const apple = V.cardRow(card({ key: 'apple::Apple M5 Max', name: 'Apple M5 Max', vendor: 'apple', kind: 'apple', worker: 'Metal', vram_mb: 131072, power_w: 0, power_default_w: 0, temp_gpu: 0, temp_mem: 0, telemetry_at: 0, hash_now: 7.4, accepted: 0, prepared: false }));
assert.equal(apple.kindWord, 'Apple silicon');
assert.equal(apple.hash, '7.4');
assert.equal(apple.temp, '');
assert.equal(apple.power, '');
assert.equal(apple.cap, 0);
assert.equal(apple.meta, '128 GB unified');
});
test('an integrated GPU is shown as integrated and off, with its reason', () => {
const r = V.cardRow(card({ key: 'intel:1:UHD', name: 'Intel UHD Graphics 770', vendor: 'other', kind: 'integrated', enabled: false, state: 'off', hash_now: 0, reason: 'integrated GPU: slow and shares the machine memory', power_w: 0, temp_gpu: 0 }));
assert.equal(r.integrated, true);
assert.equal(r.kindWord, 'Integrated');
assert.equal(r.on, false);
assert.equal(r.word, 'off');
assert.equal(r.tone, 'off');
assert.equal(r.hash, '');
assert.equal(r.sub, 'integrated GPU: slow and shares the machine memory');
assert.equal(V.cardRow(card({ kind: 'unknown' })).canToggle, false);
assert.equal(V.cardRow(card({ state: 'restarting', restart_in_s: 7 })).word, 'restart in 7 s');
assert.equal(V.cardRow(card({ state: 'faulted' })).tone, 'bad');
assert.equal(V.cardRow(card({ state: 'waiting' })).word, 'waiting for the node');
});
test('the big button: start when paused, stop when mining, disabled with no card on', () => {
const m = (over) => ({ state: 'mining', paused: false, cards: [card()], ...over });
assert.deepEqual(V.toggle(m(), { synced: true }, {}), { label: 'Stop mining', sub: 'mining on 1 of 1 card', cls: 'stop', disabled: false, act: 'pause' });
assert.deepEqual(V.toggle(m({ state: 'paused', paused: true }), { synced: true }, {}), { label: 'Start mining', sub: 'paused · the node keeps running', cls: '', disabled: false, act: 'resume' });
const none = V.toggle(m({ cards: [card({ enabled: false })] }), { synced: true }, {});
assert.equal(none.disabled, true);
assert.equal(none.act, '');
assert.equal(none.sub, 'switch a GPU on below first');
assert.equal(V.toggle(m({ cards: [] }), { synced: true }, {}).sub, 'no GPU this app can drive');
assert.equal(V.toggle(m({ state: 'waiting' }), { synced: false }, {}).sub, 'waiting for the node to sync');
assert.equal(V.toggle(m(), { synced: true }, { quitting: true }).disabled, true);
});
test('the node words: synced, syncing with a percentage, failed, a blocked clock', () => {
const n = (over) => ({ state: 'synced', blocks: 135200, headers: 135200, peers: 3, daa: 140000, last_reading_age_s: 4, message: '', restart_in_s: 0, ...over });
const ok = V.nodeWords(n(), { severity: 'none' }, '');
assert.equal(ok.word, 'synced');
assert.equal(ok.tone, 'ok');
assert.match(ok.line, /every block/);
const sy = V.nodeWords(n({ state: 'syncing', blocks: 50000, headers: 100000 }), { severity: 'none' }, 'about 3 min left at 300 blocks/s');
assert.equal(sy.word, 'syncing');
assert.equal(sy.line, 'about 3 min left at 300 blocks/s · 50,000 of 100,000 (50%)');
assert.equal(V.nodeWords(n({ state: 'failed', message: 'igneumd could not start' }), { severity: 'none' }, '').tone, 'bad');
const blocked = V.nodeWords(n(), { severity: 'block', skew_s: -75 }, '');
assert.equal(blocked.tone, 'bad');
assert.match(blocked.line, /clock is off by 75 s/);
assert.equal(V.peersLine(n({ peers: 0 })), 'none yet: looking for the seed node');
assert.equal(V.peersLine(n({ peers: 1 })), 'one other node this one talks to');
assert.equal(V.heightLine(n({ state: 'syncing', blocks: 10, headers: 500 })), 'of 500 headers seen');
assert.equal(V.heightLine(n()), 'blocks this node holds');
});
test('the next consensus switch is the first height above the DAA score, in plain words', () => {
const sw = [
{ key: 'fees_v1_activation_daa', name: 'Fees v1', daa: 210000 },
{ key: 'difficulty_v2_activation_daa', name: 'Difficulty v2', daa: 33000 },
{ key: 'finality_v3_activation_daa', name: 'Finality v3', daa: 135200 }
];
const next = V.nextSwitch(sw, 140000);
assert.equal(next.name, 'Fees v1');
assert.equal(next.away, 70000);
assert.equal(V.switchLine(next, 140000), 'Fees v1 at DAA 210,000: 70,000 blocks away, about 19 h 27 min at one block a second.');
assert.equal(V.nextSwitch(sw, 20000).name, 'Difficulty v2');
assert.equal(V.nextSwitch(sw, 300000), null);
assert.match(V.switchLine(null, 300000), /Every planned switch is behind this node/);
assert.match(V.switchLine(null, 0), /^A switch is a planned rule change/);
assert.equal(V.nextSwitch([], 5), null);
});
test('the prove words follow the switch, the setup, the node and the status', () => {
const pv = (over) => ({ enabled: true, available: true, setup_hint: '', backend: 'cuda', status: 'idle', message: '', current: '', verifier_mode: 'command', pool_entries: 4, pool_verified: 4, pool_failed: 0, ...over });
assert.equal(V.proveWords(pv(), false, true).word, 'off');
assert.equal(V.proveWords(pv({ status: 'setup', available: false, setup_hint: 'proving needs the WSL2 setup' }), true, true).word, 'needs setup');
assert.equal(V.proveWords(pv(), true, false).word, 'waiting');
assert.equal(V.proveWords(pv({ status: 'proving', current: 'block 59199 shard 0' }), true, true).sub, 'block 59199 shard 0');
assert.equal(V.proveWords(pv({ status: 'submitted' }), true, true).tone, 'on');
assert.equal(V.proveWords(pv(), true, true).word, 'idle');
assert.equal(V.verifierWords(pv()).word, 'verifying · pool 4/4');
assert.equal(V.verifierWords(pv()).tone, 'ok');
assert.equal(V.verifierWords(pv({ verifier_mode: 'off', pool_entries: 0 })).tone, 'warn');
assert.equal(V.verifierWords(pv({ verifier_mode: 'off', verifier_reason: 'the WSL2 prover is not installed' })).needsSetup, true);
assert.equal(V.verifierWords({}).word, 'not read yet');
});
test('the dev-fee lines name the share and where Settings turns it off', () => {
const s = (on, fee) => ({ settings: { dev_fee: on }, mining: { fee_total: fee }, dev_fee: { on, percent: on ? 1 : 0, address: '0x1234567890abcdef1234567890abcdef12345678', line: '' } });
assert.equal(V.devFeeLine(s(true, 12)), 'One block in 100 pays the miner software’s dev fee (12 so far). Settings turns it off.');
assert.equal(V.devFeeLine(s(false, 0)), 'The dev fee is off. Every block pays this address.');
assert.equal(V.devFeeText(s(true, 0)), 'Dev fee 1% (1 block in 100) to 0x123456…5678');
assert.match(V.devFeeText(s(false, 0)), /^Dev fee off/);
});
test('the remote-jobs line and the helpers', () => {
const title = (j) => j.title || j.kind;
assert.equal(V.jobsNote({ allowed: false }, 1000, title), 'Off: nothing runs here until Settings allows remote jobs.');
assert.equal(V.jobsNote({ allowed: true, url_set: false }, 1000, title), 'No jobs address in this build.');
assert.equal(V.jobsNote({ allowed: true, url_set: true, active: true, id: 'job-3', kind: 'build', title: 'build' }, 1000, title), 'Running build (job-3).');
assert.equal(V.jobsNote({ allowed: true, url_set: true, checked_at: 940, queued: 2 }, 1000, title), 'Nothing running; checked 1 min ago; 2 queued.');
assert.equal(V.shortHex('0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a'), '0x2b1a81…79ef7a');
assert.equal(V.shortHex('0xabc'), '0xabc');
assert.equal(V.withCommas(1234567), '1,234,567');
assert.equal(V.compact(2500000), '2.50M');
assert.equal(V.rel(90), '1 min ago');
});
test('hot-plug (src/hotplug.rs): a removed card and a faulty card are shown as such, with no switch, and leave the counts', () => {
const gone = card({ key: 'amd:gfx1201', name: 'gfx1201', vendor: 'amd', state: 'removed', removed_at: 1000, added_at: 0, gone: false, hash_now: 0 });
const r = V.cardRow(gone);
assert.equal(r.removed, true);
assert.equal(r.on, false);
assert.equal(r.canToggle, false);
assert.equal(r.word, 'removed');
assert.equal(r.tone, 'off');
assert.equal(r.hash, '');
assert.match(r.sub, /^unplugged; its worker stopped/);
const bad = card({ key: 'amd:gfx1201#2', name: 'AMD Radeon RX 9070 XT', code: 'gfx1201', vendor: 'amd', enabled: false, state: 'unusable', problem: 'Code 43', message: 'not usable (Code 43)', reason: 'reboot with the card attached; if it persists, reinstall the driver with the card attached', added_at: 900, removed_at: 0, device: '1', platform: 'AMD Accelerated Parallel Processing', bus: '0000:03:00.0' });
const b = V.cardRow(bad);
assert.equal(b.unusable, true);
assert.equal(b.canToggle, false);
assert.equal(b.word, 'not usable (Code 43)');
assert.equal(b.tone, 'bad');
assert.match(b.sub, /^reboot with the card attached/);
assert.equal(b.title, 'gfx1201 · CUDA device 1 · AMD Accelerated Parallel Processing · bus 0000:03:00.0');
assert.equal(V.cardRow(card()).title, 'CUDA device undefined'.replace(' device undefined', '') === '' ? '' : V.cardRow(card()).title);
// a row that is gone (five minutes after removal) is not shown at all; the big button counts only present cards
assert.deepEqual(V.shownCards([card(), card({ key: 'x', gone: true })]).map((c) => c.key), ['nvidia:0:RTX 5090']);
assert.equal(V.present(card()), true);
// performance order (6 October 2026): PC 1 detects 5090, integrated AMD, 9070 XT; the list shows the 5090, the 9070 XT,
// then the integrated card, whatever the detection order and whatever the integrated card's rate
const pc1 = [
card({ key: 'nvidia:0:RTX 5090', hash_avg: 122 }),
card({ key: 'amd:gfx1036', name: 'AMD Radeon(TM) Graphics', vendor: 'amd', kind: 'integrated', vram_mb: 512, hash_avg: 2.1, enabled: true }),
card({ key: 'amd:gfx1201', name: 'AMD Radeon RX 9070 XT', vendor: 'amd', kind: 'discrete', vram_mb: 16384, hash_avg: 18.2 }),
];
assert.deepEqual(V.shownCards(pc1).map((c) => c.key), ['nvidia:0:RTX 5090', 'amd:gfx1201', 'amd:gfx1036']);
// before any rate (first start): discrete by memory, integrated last; a removed card last of all; ties keep detection order
const fresh = [
card({ key: 'amd:gfx1036', kind: 'integrated', vram_mb: 512, hash_avg: 0, hash_now: 0, state: 'off' }),
card({ key: 'amd:gfx1201', kind: 'discrete', vram_mb: 16384, hash_avg: 0, hash_now: 0, state: 'waiting' }),
card({ key: 'nvidia:0:RTX 5090', hash_avg: 0, hash_now: 0, state: 'waiting' }),
card({ key: 'nvidia:1:RTX 5090', hash_avg: 0, hash_now: 0, state: 'waiting', removed_at: 5 }),
];
assert.deepEqual(V.shownCards(fresh).map((c) => c.key), ['nvidia:0:RTX 5090', 'amd:gfx1201', 'amd:gfx1036', 'nvidia:1:RTX 5090']);
// a small rate change does not reorder (5 MH/s buckets): 122 and 124 sort as equal and keep detection order
assert.deepEqual(V.shownCards([card({ key: 'a', hash_avg: 122 }), card({ key: 'b', hash_avg: 124 })]).map((c) => c.key), ['a', 'b']);
assert.equal(V.present(gone), false);
assert.equal(V.present(bad), false);
const t = V.toggle({ state: 'mining', paused: false, cards: [gone, bad] }, { synced: true }, {});
assert.equal(t.disabled, true);
assert.equal(t.sub, 'no GPU this app can drive');
const t2 = V.toggle({ state: 'mining', paused: false, cards: [card(), gone] }, { synced: true }, {});
assert.equal(t2.sub, 'mining on 1 of 1 card');
});

View file

@ -116,7 +116,7 @@ final class App: NSObject, NSApplicationDelegate, WKNavigationDelegate, WKUIDele
window.titleVisibility = .hidden
window.isMovableByWindowBackground = true
window.backgroundColor = obsidian
window.minSize = NSSize(width: 900, height: 620)
window.minSize = NSSize(width: 900, height: 600)
window.center()
window.delegate = self
window.isReleasedWhenClosed = false

View file

@ -16,6 +16,7 @@
#endif
#include <windows.h>
#include <shellapi.h>
#include <dbt.h>
#include <wrl.h>
#include <string>
#include <vector>
@ -284,6 +285,9 @@ static LRESULT CALLBACK WndProc(HWND hwnd, UINT msg, WPARAM wp, LPARAM lp) {
case WM_SIZE:
if (g_controller) { RECT rc; GetClientRect(hwnd, &rc); g_controller->put_Bounds(rc); }
return 0;
case WM_GETMINMAXINFO: // the dashboard lays out from 900 x 600 up (app/igneum-app/ui); the Mac window says the same
((MINMAXINFO*)lp)->ptMinTrackSize.x = 900; ((MINMAXINFO*)lp)->ptMinTrackSize.y = 600;
return 0;
case WM_ENGINE_LINE: {
if (wp == 1) {
g_exited = true;
@ -324,6 +328,11 @@ static LRESULT CALLBACK WndProc(HWND hwnd, UINT msg, WPARAM wp, LPARAM lp) {
}
}
return 0;
case WM_DEVICECHANGE:
// a device arrived or left (an eGPU through a USB4 box, a driver coming up or crashing): the engine enumerates
// the cards now instead of at its next minute poll (src/hotplug.rs); DBT_DEVNODES_CHANGED needs no registration
if (wp == DBT_DEVNODES_CHANGED || wp == DBT_DEVICEARRIVAL || wp == DBT_DEVICEREMOVECOMPLETE) sendEngine("detect");
return TRUE;
case WM_TRAY:
if (lp == WM_LBUTTONUP || lp == WM_LBUTTONDBLCLK) { ShowWindow(hwnd, SW_SHOW); SetForegroundWindow(hwnd); }
else if (lp == WM_RBUTTONUP || lp == WM_CONTEXTMENU) showTrayMenu();

View file

@ -3,6 +3,6 @@
// packaging/windows/Igneum-Miner.iss when the app version moves. Include guards, not #pragma once: rc.exe reads it too.
#ifndef IGNEUM_HOST_VERSION_H
#define IGNEUM_HOST_VERSION_H
#define IGNEUM_HOST_VERSION_STR "0.3.9"
#define IGNEUM_HOST_VERSION_RC 0,3,9,0
#define IGNEUM_HOST_VERSION_STR "0.3.12"
#define IGNEUM_HOST_VERSION_RC 0,3,12,0
#endif

View file

@ -0,0 +1,131 @@
# Proving on AMD and Apple cards: what exists, what the CPU can do, what to tell the public
5 October 2026, from the project lead's two questions that evening: "test proving on the amd card?" and "can we test proving on
mac?". PC 1 holds an RTX 5090 and an RX 9070 XT (gfx1201, 16 GB) in an eGPU; this Mac is an M5 Max. The prover is
SP1 (`proving/igneum-prove`, `docs/plans/proving-v0.md`, `proving-v1.md`), run on the GPU only through SP1's CUDA
server. Every figure below is measured (with its bench-log entry or job id) or cited (with its file or page); the
rest is labelled approximate. Status words follow `docs/spec/00-overview.md` 0.2.
**The answer in three lines.** No zkVM proves on an AMD GPU on 5 October 2026: not SP1, not RISC Zero, not Jolt, not
OpenVM, and the ICICLE library underneath them has no AMD backend either. Apple silicon has a shipped Metal prover in
RISC Zero and a Metal backend in ICICLE, but SP1, the prover Igneum runs, is CPU-only on a Mac. So an AMD-only or
Apple-only machine mines and does not prove on its card; it can prove on its CPU, at the times measured in section 2.
## 1. The backends (read 5 October 2026, 20:30 to 20:50 UTC)
| Prover | Version read | CPU | NVIDIA (CUDA) | AMD (ROCm or HIP) | Apple (Metal) | Vulkan or WebGPU | Where it says so |
|---|---|---|---|---|---|---|---|
| SP1 (ours) | v6.8.1, 24 Sep 2026 (pinned); `dev` head 318dd530, 28 Sep 2026 | yes; AVX2 and AVX-512 on x86 through Plonky3 | yes: `sp1-gpu-server`, "Compute Capability 8.0 or higher", "24GB or more VRAM", "the CUDA 12 runtime and a compatible NVIDIA driver", Linux x86_64 | **no** | **no** | **no** | docs.succinct.xyz, SP1 docs "Hardware acceleration" page; `crates/sdk/src/lib.rs` (`pub mod cpu`, `mock`, `light`, `#[cfg(feature = "cuda")] pub mod cuda`, `#[cfg(feature = "network")] pub mod network`: no other backend module); `sp1-gpu/README.md` (`CUDA_ARCHS` 89, 90, 100, 120; NTT by NVIDIA cuPQC or sppark); release notes v6.2.3 to v6.8.1 (the only backend line: "add optional cuPQC NTT backend", v6.8.0); a code search of the repository on 5 October: "rocm" 0 files, "metal" 0, "vulkan" 0, "webgpu" 0; `cuobjdump` of sp1-gpu-server 6.8.1: sm_80, 86, 89, 90, 100, 120 and compute_120 PTX, nothing else (`docs/bench-log.md`, "proving v1", 5 October 2026) |
| sppark (SP1's NTT fallback, vendored at `sp1-gpu/crates/sys/sppark`) | `main` README, read 5 October 2026 | | yes: "x86_64 with Nvidia's Volta+ GPU hardware platforms on Linux and Windows" | "A limited support for AMD's RDNA and CDNA GPUs is provided" (upstream README). SP1's tree carries no HIP build: the 0 "rocm" files above, and `sp1-gpu/crates/sys/sppark/util/gpu_t.cuh` is CUDA only | no | no | github.com/supranational/sppark README; the SP1 files named |
| RISC Zero | latest release v3.0.6, 17 Jul 2026 (a v5.0.0-rc.1 of 15 Jan 2026 is also on the releases page) | yes, "nearly any modern CPU (x86 or ARM)" | yes, "RISC Zero targets NVIDIA GPUs using the CUDA framework" | **no** ("rocm", "vulkan": 0 files in the repository) | **yes**: `metal = ["prove"]` in `risc0/zkvm/Cargo.toml`; kernels in `risc0/sys/kernels/zkp/metal/*.metal` (zk, fri, mix, sha); docs: "RISC Zero will use the integrated Metal compute cores" on Apple silicon. The Groth16 wrapper "only works on x86 architecture, and so Apple Silicon is currently unsupported (even via Docker)" | no | dev.risczero.com "Local proving"; `risc0/zkvm/Cargo.toml` features `cuda = [... risc0-zkp/cuda ...]`, `metal = ["prove"]` |
| Jolt (a16z) | v0.3.0-alpha, 1 Oct 2025; "Jolt is in alpha and is not suitable for production use" | yes, "state-of-the-art performance on CPU" | no | **no** | a **draft** PR #1733 (opened 3 Aug 2026, not merged): titled "feat: Metal GPU backend (Apple Silicon)" and marked experimental: 92 Metal kernels, "2.12x speedup at 2^20 scale" on an M4 mini and "3.20x vs same-binary CPU" on an M5 Max, "Apple Silicon + macOS only. No CI coverage" | no | github.com/a16z/jolt README and book (jolt.a16zcrypto.com); PR #1733 |
| OpenVM | v2.0.2, 14 Aug 2026 | yes | yes: `cuda-backend` (v1.4.2 notes), "Improves the Halo2 GPU prover" (v2.0.2) | **no** | **no** | no | github.com/openvm-org/openvm releases |
| ICICLE (Ingonyama; the GPU library behind several provers, not SP1) | v4.0.0, 11 Jul 2025 | yes (MIT) | yes, "CUDA (for NVIDIA GPUs)" | **no** backend listed | yes, "Metal (for Apple Silicon GPUs)"; both under a special licence with a free research licence | Vulkan in the build system (PR #735, merged Jan 2025) and a draft "Vulkan NTT" PR #1019 (Jul 2025, "still wip"); nothing installable | dev.ingonyama.com "Install GPU backend"; the releases page; the README ("backends ... are distributed under a special license") |
Said plainly: **on 5 October 2026 no zkVM proves on an AMD GPU.** The only AMD code in the whole chain is sppark's
limited HIP path, which SP1 does not build. For Apple silicon the answer is split: RISC Zero ships a Metal prover
and ICICLE a Metal backend; SP1, Jolt and OpenVM do not. Nothing read tonight names an AMD plan with a date.
## 2. The CPU fallback, measured
SP1's CPU prover is the path an AMD-only or Apple-only machine has today. Three machines, the same pinned guests
(shard program id `0x2b1a81cb...`, aggregator `0x474678f3...`, pinned 2026-10-05T16:20:38Z), `SP1_PROVER=cpu`,
`--mode shard --shard 0` (execute, core proof, compressed proof, each verified). The RTX 5090 rows are the reference.
| Fixture (SP1 cycles) | Stage | PC 1 CPU, miner running on both cards (job `cpu-prove-pc1-small2`) | Apple M5 Max CPU (4 October, loaded; bench-log) | RTX 5090 (bench-log) |
|---|---|---|---|---|
| block-56-transfers-3shards shard 0, 200 pgas (315 k) | core | 82.5 s, 7,310,257 B, verify 0.210 s | 83.1 s, 7,310,257 B | not run on the 5090; the nearest rows are block-78 below and an empty live shard: 7.0 to 7.7 s compressed with the miner on the card (5 Oct, `chain-pc2-pv1b`, `pv1c`) |
| | compressed | 199.2 s, 1,272,897 B, verify 0.035 s; 312 s wall for setup 22.8 s, execute 0.14 s, core, compressed | 272.3 s, 1,272,897 B | |
| | peak RSS, CPU | 29.5 GB peak RSS; 978% CPU (9.8 of 16 cores), user 2,516 s, system 537 s | not recorded | |
| block-78-increment, 2 transactions (626 k) | core | 87.0 s, 7,317,857 B, verify 0.209 s | 22.0 s, 7.3 MB (3 October, v0 guest) | 1.4 s (4 October, mining paused) |
| | compressed | 202.3 s, 1,272,897 B, verify 0.034 s; 322 s wall (setup 21.8 s) | 55.7 s, 1.27 MB | 2.7 s |
| | peak RSS, CPU | 30.5 GB peak RSS; 979% CPU, user 2,616 s, system 541 s | not recorded | |
| block-338-shard1, one shard at `S_p` (60.8 M) | core | **not run**, by the PC 1 scheduler's decision at 21:05Z (PC 1's time tonight belongs to the Counter ASIC 2.0 gates; the job `cpu-prove-pc1-sp`, script `tools/amd-prove/pc1-cpu-prove-sp.ps1`, is written and unpublished). Extrapolation, approximate: 60.8 M cycles is about 29 SP1 shards of 2^21 cycles where the small fixtures are one, so the core proof alone is about 29 x 80 s, 40 min, and the compressed recursion over 29 shard proofs adds hours; the floor from the 5090's own ratios (6x on core, 4x on compressed between block-78 and `S_p`) is 9 min core and 13 min compressed. Either way far outside every deadline | not run on the CPU (execute alone 6.9 s) | 8.3 s |
| | compressed | not run (see the core cell) | not run | 10.9 s with the card to itself (4 Oct); 33.0 s with the miner running (5 Oct, `memminer-pc2-pv1`); 7.3 to 7.7 s per EMPTY shard with the miner running (`chain-pc2-pv1c`) |
| | peak RSS, CPU | not run; at least the 30 GB of the small rows | | GPU peak 28,295 MiB alone, 30,039 MiB beside the miner |
PC 1: Windows 11, WSL2 Ubuntu 24.04 as root, 16 cores and 46,994 MB visible to the VM, the Igneum Miner app 0.3.9 mining on the RTX 5090 (89% mean utilisation through both runs, 59 to 70% minimum: the miner, untouched) and on the RX 9070 XT (not visible to nvidia-smi, mining through the app's OpenCL worker). Job `cpu-prove-pc1-small2`, 20:49:00Z to 20:59:49Z, 649 s wall including a 6-s warm build; the host built without the `cuda` feature from the hosted package `igneum-prove-wsl2-pv1b.zip`, `--mode id` the pinned pair. The first job, `cpu-prove-pc1-small` (20:44 to 20:46Z), built the host cold in 126 s and proved nothing: an apostrophe inside a single-quoted awk program ended the quote, bash refused the whole loop and the job reported exit 0. The class fix: `tools/amd-prove/check-job-bash.sh` runs `bash -n` on the bash body of a PowerShell job before it is published, and the job itself runs `bash -n` inside the distro before the run; both were shown to fire on the bad body and pass the fixed one. Host RAM in the VM: 968 MB used before, 2,351 MB after; the prover's own peak 29.5 to 30.5 GB.
Mac, fresh run tonight: not taken. The Mac measure lock was held from 20:31Z (a read-width `packbench` under `measure`, three build slots, then a 1,500-s proving-v1 network under `run`) and did not free inside the 10-minute window the coordinator set, so the Apple column is the 4 October rows (M5 Max, 18 cores, 64 GB, load 38 to 47, `nice -n 19`): the same host modes on the same fixture, under heavier load than PC 1 tonight. The Mac's RAM peak was not recorded on 4 October; PC 1's 30 GB says a Mac needs more than 32 GB for the CPU prover, which a 64 GB M5 Max has and a 16 or 24 GB Mac does not.
The deadlines a CPU proof has to fit (all in the spec and the v1 plan): the exclusive window of an assigned shard is
10 s of DAA time (spec 7.2 item 3; after it anyone may prove and be paid first); the litepaper promises the block's
proof "within about a minute"; the launch target is 20 to 60 s behind the tip; from proving v1 a segment nobody has
proven in `T` = 600 DAA s (10 min) pays nothing (`docs/plans/proving-v1.md`, decisions). So a CPU shard proof is
useful only if it lands inside 10 min and competitive only if it lands inside about a minute.
Reading. On PC 1 the CPU proof of the smallest shard (315 k cycles) and of the two-transaction block (631 k cycles) cost the same: 82.5 and 87.0 s core, 199.2 and 202.3 s compressed. Doubling the cycles added 4.5 s to the core proof and 3.1 s to the compressed one, so about 280 s of every CPU proof is fixed cost (the recursion that turns the core proof into the 1.27 MB compressed proof the chain carries), and no shard size removes it. Against the deadlines: 282 s a shard (core plus compressed, the client already set up, as the app's loop runs it) is 28x the 10-s assignment window, 4.7x the minute the litepaper promises, and inside the 600-s unproven deadline of v1 with 5 min to spare; but an NVIDIA card proves the same shard in 2.7 to 7.7 s, so a CPU prover only ever wins a shard that no card has taken in 10 minutes. The Mac's 83.1 and 272.3 s of 4 October have the same shape. The RAM peak of 29.5 to 30.5 GB is the second finding: the SP1 CPU prover does not fit a 16 GB machine at all, and WSL2 gives a Windows VM half the host's RAM by default, so the CPU path needs a 64 GB Windows PC or a 32 GB Linux or Mac machine. The `S_p` shard on the CPU can only be slower (the 5090 takes 6x longer at `S_p` than on block-78: 8.3 s against 1.4 s core); it was not run tonight (the scheduler kept PC 1 for the Counter ASIC 2.0 gates) and could not change the conclusion.
## 3. What this means for each tier (the every-number rule, CLAUDE.md 5 October 2026)
| Tier | Mines | Proves on the card | The 20% proving-pool share (spec 2.5) | What the software does today |
|---|---|---|---|---|
| AMD-only home miner, one card of 8, 12 or 16 GB (an RX 9070 XT is 16 GB), Windows or Linux | yes (OpenCL worker, `proto-opencl`; PC 1's 9070 XT mines on the devnet) | **no**: no prover exists for the card | **lost**, unless CPU proving at a small shard size becomes a tier (section 4a) | the rig installer: `prover_decision` in `packaging/linux/bin/igneum-rig-lib.sh` (branch `rig-install`) skips every non-NVIDIA card (`[[ "$vendor" == nvidia ]] \|\| continue`) and prints "proving off by default: no NVIDIA card (no CUDA prover for AMD or Intel yet)"; the app: `provedefault.rs` (branch `proving-v1`) considers NVIDIA cards only. Both already right; neither offers the CPU path |
| Apple silicon (M-series, unified memory) | yes: the M5 Max at 26.7 MH/s (bench-log 4 October, "first hourly program swap", Metal `prepare 1` row) | **no** with SP1; RISC Zero and ICICLE have Metal, SP1 does not | **lost** today; a Metal prover behind the swappable interface would restore it (section 4b) | `provedefault.rs`: "proving stays off on Apple silicon: the M5 Max CPU took 41 to 55 s for an empty shard and minutes for a full one; Settings switches it on (CPU, slow)". Right |
| Mixed rig (NVIDIA and AMD cards in one box) | every card | the NVIDIA cards prove for the box; the AMD cards mine | kept, earned by the NVIDIA cards | the rig installer picks the biggest NVIDIA card (`prover_decision`, `PROVER_CARD` overrides), pauses its miner under 20 GB, keeps it mining at 20 GB or more; the AMD cards get a miner unit each. **The prover unit must never select an AMD card**: it does not (the vendor filter above), and that filter is now a stated requirement, not an accident |
| NVIDIA home miner, 8 or 12 GB | yes | no on this SP1 build (13.9 GB floor on an empty shard, `memsweep-pc2-pv1`) | lost unless the shard size moves | unchanged from `proving-v1.md` |
| NVIDIA 16 GB | yes | prove-only, miner paused per shard | kept | unchanged |
| NVIDIA 24 or 32 GB | yes | mines and proves (peak 16.8 GB on empty shards, 30.0 GB on a full prototype shard beside the miner) | kept | unchanged |
| Pool user | through the pool | the pool's own NVIDIA cards prove the shards assigned to the pool's keys (approximate: the pool protocol, spec 09, does not yet say who proves) | by the pool's rules | open, spec 09 |
## 4. The options
### 4a. CPU proving at a small shard size, as a tier
What it is: an AMD-only or Apple machine proves shards cut at a smaller budget than `S_p` on its CPU, through the
same host (`SP1_PROVER=cpu`; the host's `--budget` re-plan from branch `proving-v1`, commit c2544be, cuts a fixture at
any budget). The miner keeps the card; the prover takes the CPU.
What the numbers say: the fixed cost kills it. 282 s a shard on a 16-core PC and 355 s on the loaded M5 Max, with 30 GB of RAM, at the smallest shard there is; the time sits in the compressed-proof recursion, not in the cycles, so cutting shards smaller does not help, and the launch deadline (20 to 60 s behind the tip) is missed by 5x. It fits only the v1 unproven deadline (600 s), which pays a CPU prover only when no card has proven the shard in 10 minutes: on a chain with one NVIDIA prover that never happens. Recommendation: **no CPU tier**. Settings may still switch the CPU prover on (it does on macOS today), and the Proving tile must then say the proof takes about five minutes and is paid only when no card proves first.
What it costs the chain: a block cut into more, smaller shards costs more aggregation work (the aggregator guest
verifies one deferred proof per shard; 1.66 M cycles for four shards on the executor, bench-log 4 October; the
chained aggregation is 9.6 to 9.7 s per block on a mining 5090, `chain-pc2-pv1c`) and more records; the assignment
rule (8 assignees, 10 s window, spec 7.2) would need a CPU class with a longer window or the CPU provers only ever
win the open phase. None of that is measured. Status: Designed, nothing implemented.
### 4b. A second prover backend behind the swappable interface
The seam exists: `proving/igneum-prove/host/src/proof_system.rs` (`ProofSystem` trait, `Sp1ProofSystem`,
`StubProofSystem`), versioned per the design. The candidates:
| Target | Most likely backend | What exists | What adopting it costs |
|---|---|---|---|
| Apple silicon | RISC Zero's Metal prover (`metal` feature, shipped) | a shipped feature with kernels in the tree; ICICLE's Metal backend as the other library | a second guest program (the shard statement, `core/` is plain Rust and ports; the precompile patches for keccak and secp256k1 differ), a second pinned program id and verifying key in `elf/manifest.json`, the node's verifier for both proof formats (RISC Zero receipt and SP1 compressed proof) in `--mode verify` and the proof pool, and an aggregation problem: SP1's aggregator folds SP1 proofs by deferred verification; it cannot fold a RISC Zero receipt, so a block with shards from both families needs two aggregations or a wrapper. Approximate: weeks of a person's time, no measurement of a Metal shard time exists; RISC Zero's Groth16 wrapper for light clients does not run on Apple silicon at all |
| AMD | nothing | sppark's limited HIP path (not in SP1's tree); ICICLE's and Jolt's Vulkan and Metal work are not AMD | no backend to adopt. The honest statement is that it lands when a zkVM ships one |
### 4c. The public line
The site says today (read 5 October 2026 from `site/litepaper.html`, `site/miner.html`, `site/index.html`): "The same
card proves every block", "The card mines and proves", "Ember finds your GPU, makes a wallet for you and runs the
node, the miner and the prover as one app", "Target: shard size will be set so a 12 GB card proves one shard in about
20 seconds". Every one of those is true of an NVIDIA card with enough memory and false of an AMD or Apple card, and the
litepaper's own rule is "If consumer GPUs cannot prove shards fast enough, Igneum says so and does not launch on promises".
The recommended line, for the litepaper's proving section, the miner page and the app's Proving tile (copy law):
> Proving needs an NVIDIA card with 16 GB or more today (20 GB to mine and prove on the same card). AMD and Apple
> cards mine. A prover for them lands when a zkVM ships one. A CPU can prove a small shard in about five minutes with 32 GB of RAM free; the chain pays the first proof, which a card delivers in seconds, so CPU proving is for testing, not income.
Where the numbers come from: 16 GB and 20 GB are the measured gates of `proving-v1.md` (13.8 GB prover-alone peak,
16.8 GB mine-and-prove peak); "when a zkVM ships one" is section 1. The line changes when the memory sweep moves the
gates or a backend ships; it is reviewed with every prover release.
## 5. What this analysis does about it (the consequences, before anyone asks)
| Consequence | Action | Owner |
|---|---|---|
| An AMD-only miner loses the proving share | the CPU tier of 4a is measured here (section 2); whether it becomes a tier is a decision for the project lead on those numbers | this analysis; the project lead |
| The rig's prover unit must select NVIDIA cards only | already true in `prover_decision`; told the rig-installer agent to keep it as a stated rule and to print the CPU-fallback line for AMD-only rigs | rig-installer agent |
| The app's Proving tile on an AMD-only or Apple machine should say why it is off and name the CPU path | the `provedefault.rs` lines already say so for Apple; AMD-only Windows machines get "no NVIDIA card ..." | proving agent (told) |
| The site and litepaper over-promise for AMD and Apple | the line of 4c, to land with the next site pass (copy law; `node site/build.mjs`; link-check) | site-pages owner; not changed here |
| A Metal prover is the only non-NVIDIA path with a shipped backend | 4b names RISC Zero's Metal path and its cost; no work started | proving agent (told) |
## 6. Commands, jobs and sources
| What | Where |
|---|---|
| The PC 1 jobs (signed `run` jobs, PowerShell, not elevated, miners untouched, SP1_PROVER=cpu, host built without the `cuda` feature) | `tools/amd-prove/pc1-cpu-prove.ps1` (small fixtures), `pc1-cpu-prove-sp.ps1` (the `S_p` shard); published as `cpu-prove-pc1-small` (built, proved nothing: the quote bug) and `cpu-prove-pc1-small2` (the numbers) by `packaging/ota/publish-jobs.sh add --kind run --target ae432dc7 --shell powershell --timeout-minutes 60`; `cpu-prove-pc1-sp` written, not published; `check-job-bash.sh` gates the bash body of every job script here; the package `igneum-prove-wsl2-pv1b.zip` (sha256 df50dee5...), the URL and hash filled at publish time, never committed |
| The Mac run | `tools/lock/with-lock.sh measure /usr/bin/time -l igneum-prove-host block-56-transfers-3shards.json --mode shard --shard 0` with the `proving-v1` worktree's host (pinned ids checked with `--mode id`) |
| Results | `node tools/jobs.mjs cpu-prove-pc1-small2 --all`, `docs/bench-log.md` entry "5 October 2026, the CPU prover on PC 1 and the backend survey" |
| Pages read | SP1: docs.succinct.xyz hardware-acceleration page, github.com/succinctlabs/sp1 (releases, `crates/sdk/src/lib.rs`, `sp1-gpu/README.md`, code search); RISC Zero: dev.risczero.com local-proving, `risc0/zkvm/Cargo.toml`; Jolt: README, book, PR #1733; OpenVM: releases; ICICLE: install_gpu_backend page, releases, README, PRs #735 and #1019; sppark README |

View file

@ -0,0 +1,418 @@
# ASIC resistance, 2011 to 2026: the history, the papers, the lessons, and the audit of Igneum against them
5 October 2026 (night), branch `asic-history`. Asked by the project lead at 20:05 UTC: "do a full on deep dive into the full history of 'asic resistance' and see if we can add or upgrade anything." Baseline for the audit: the Counter ASIC 2.0 final class decided tonight (`docs/plans/counter-asic-2-status.md` on `ca2-coord`, entries 20:16 to 22:25 UTC; `docs/analysis/chip-model-v3.md` on `ca2-mixer` 1ab8b21). Every figure about another chain cites a repo file, a paper or a dated article, or is labelled approximate. Hash-per-joule gains are computed from the cited hashrate and watt figures of the chip and of the best consumer GPU of the same year, and are approximate by construction (GPU figures vary by tuning). Research gathered by four sub-agents between 20:10 and 20:45 UTC; the fetch failures they reported are listed in section 6.
## 0. One page for the project lead
**What the history says Igneum is doing right.**
| # | What | The evidence |
|---|---|---|
| 1 | Binding the hash to random reads over a dataset larger than any on-chip cache, with the dataset growing on a schedule, and measuring the latency-bound share per card | Every compute-bound hash fell to a chip at 20x to 1,200x per joule within 16 to 37 months (rows Scrypt, X11, Blake, kHeavyHash, Blake3). The memory-bound hashes capped the chip at 1.1x to 4.8x (Ethash rows) or saw no chip at all (KawPow, Verthash, Autolykos, FishHash). RandomX's own design chose a 2 GiB dataset because a 2 GiB SRAM die was "questionable" at 7 nm in 2019 (RandomX `doc/design.md`) |
| 2 | A random program per epoch from a VDF seed, a weak-program filter, era draws from chain state, instruction families unlocked by height, and no scheduled human fork | Monero forked four times in 20 months and lost 85% of its hashrate at each fork to chips that returned within months (CryptoNight rows). Vertcoin forked three times and was 51%-attacked after two of them. Ravencoin's X16Rv2 fork was followed by FPGA bitstreams within weeks. Grin's six-monthly tweaks worked only because the lane was scheduled to die. The only random-program hashes with no chip after five years are the ones that never needed a fork (KawPow, FiroPoW, ProgPowZ) |
| 3 | Pricing the on-die-cache recompute chip and spending the mixer budget against it (x8: 0.92x with the 3x factor), with the model public | This is the "light-evaluation attack" Least Authority flagged on ProgPoW in 2019 and ProgPoW never fixed; Bob Rao's hardware audit put an on-die-DAG ProgPoW chip at "<< 0.1x" the energy per hash of a GPU. Kik's 2020 exploit was the same attack through a 64-bit seed. Igneum has a number against it; ProgPoW had a suggestion |
**What the history says Igneum is missing or under-weighting.**
| # | What | The evidence |
|---|---|---|
| 1 | The partial-store chip with a custom memory system (HBM or many narrow DRAM channels) is not in the chip model. The model prices only the f = 0 endpoint (all SRAM, recompute everything) | The only chip class that ever beat a memory-bound GPU hash did it this way: Ethash chips reached 2.1x (Linzhi, 2020), 2.9x (E9, 2022) and 4.8x (Jasminer X4, 2021) per joule through custom memory controllers and on-package memory, with no on-die dataset at all. The time-memory curve between f = 0 and f = 1 is open item O-1.6 and has never been drawn (`proto-metal/MEMHARD.md` section 3 item 2). Cuckoo Cycle's "tmto-hard" claim fell to a 50x memory cut for 2x time within two months of publication (Andersen, 2014) |
| 2 | The item-derivation mixer has a fixed shape. That fixed shape is exactly what hands the recompute chip its 3x fixed-function factor (bare 0.31x becomes 0.92x) | RandomX made the item derivation itself a random program (SuperscalarHash: about 450 instructions generated per seed, scheduled for a superscalar core, 170-cycle latency to match DRAM), so a chip cannot hard-wire it. Igneum draws the mixer's constants per day and keeps the shape; a chip hard-wires the shape. CryptoNight-R's random math per block raised chip latency only 2.5x; the lever is in the memory path, which for the recompute chip is the mixer |
| 3 | No clock and no detector. Chips have appeared at market caps from $18M (Radiant) and $23M (Grin) upward, and Vorick's 2018 rule was "any coin with over $20M of block reward in a year has a secret ASIC on it". Monero's secret chips held 85% of the hashrate before anyone saw them. Igneum's bounty is unfunded (D11) and the public benchmark is January 2027 | Rows Monero, Radiant, Grin, Handshake, Kadena; section 2.5 (the market-cap table); MoneroCrusher's nonce analysis (February 2019) found the chips by their share pattern, which is the detector Igneum can run from day one on the observer |
**The one change I would make first.** Add the partial-store HBM chip to the chip model and draw the time-memory curve before genesis (ranked addition 1, section 4.3). It is an analysis, it costs the GPU nothing, and it is the one chip class that history shows beating memory-bound GPU work. If that row comes out under 2x the x8 decision stands as measured; if it does not, the next lever is known before the vectors are frozen.
Everything else in this document is the evidence behind that page.
## 1. The history, one row per attempt
Columns: what the hash relied on; when it went live; the first chip that beat it (vendor, model, date, rated hashrate and watts); the gain in hash per joule against the best consumer GPU of the time (approximate, derived); how long it held (months from live to first chip); what the chain did. "None" in the chip column means no shipped chip was found in any source as of October 2026.
### 1.1 The rows
| # | Hash (chain) | Live | Relied on | First chip (vendor, model, date, rate, watts) | Gain per joule vs GPU (approximate) | Held (months) | Response | Sources |
|---|---|---|---|---|---|---|---|---|
| 1 | Scrypt (Tenebrix, Litecoin, Dogecoin) | Sep and Oct 2011 | 128 KB scratchpad, meant to fit a CPU cache and not a 2011 GPU; latency at SRAM scale | Gridseed GC3355, early 2014, 360 kH/s at 7 to 8 W; Innosilicon A2 Terminator, Apr 2014, 28 nm; KnC Titan, 2014, 300 MH/s at 850 W | Gridseed 19x, A2 68x, Titan 140x (vs Radeon 7970 at 700 kH/s, 285 W); Antminer L7 (2021) 1,100x | 27 | Embraced. Dogecoin merge-mined with Litecoin from Sep 2014 | [S1] [S2] [S3] [S4] |
| 2 | X11 (Dash) and the chain family X13, X15, X17, Quark, Nist5, C11 | Jan 2014 | A chain of 11 SHA-3 candidates; compute only. Duffield said it was meant to replay Bitcoin's CPU to GPU to ASIC path, not to prevent it | iBeLink DM384M, Mar 2016, 384 MH/s at 715 W; Baikal Giant A900 (2016); Antminer D3, Sep 2017, 19.3 GH/s at 1,200 W | DM384M 33x, D3 1,000x (vs R9 280X at 4 MH/s, 250 W) | 26 | Embraced. Baikal's multi-algo units covered the whole family by 2017 | [S5] [S6] [S7] |
| 3 | Ethash (Ethereum) | Jul 2015 | DAG of 1 GB growing per epoch, 128-byte random reads; memory bandwidth | Antminer E3, announced Apr 2018, 180 MH/s at 800 W, 4 GB DDR3; Innosilicon A10 Pro (2020) 500 MH/s at 950 W; Linzhi Phoenix (Dec 2020) 2,733 MH/s at about 3,000 W; Jasminer X4 (Oct 2021) 2.5 GH/s at 1,200 W; Antminer E9 (Jun 2022) 2.4 GH/s at 1,920 W | E3 1.1x to 1.6x (a tuned 1080 Ti beat it per watt); A10 Pro 1.2x vs RTX 3080; Phoenix 2.1x; X4 4.8x; E9 2.9x | 32 to the first chip; about 65 to a chip over 2x | ProgPoW (EIP-1057) debated 2018 to 2020 and shelved; PoS at the Merge, 15 Sep 2022. ASIC share of hashrate stayed small (one 2018 estimate: 3%) | [S8] [S9] [S10] [S11] [S12] [S13] |
| 4 | Etchash (Ethereum Classic) | Nov 2020 (Thanos, ECIP-1099) | Ethash with the DAG cut to 2.5 GB to keep 3 to 4 GB cards mining; not an anti-chip change | Post-Merge Ethash chips moved over: Antminer E9 Pro (Feb 2023) 3.68 GH/s at 2,200 W; Jasminer X16-P (2023) 5.8 GH/s at 1,900 W | E9 Pro about 4x, X16-P about 7x vs RTX 3090 (approximate) | n/a | Embraced | [S14] [S15] |
| 5 | Equihash 200,9 (Zcash, Horizen, Pirate) | Oct 2016 | Generalised birthday problem (Wagner); 144 MB in practice; memory size, with a claimed 1,000x compute penalty for halving memory | Antminer Z9 mini, announced 3 May 2018, 10 kSol/s at 300 W; Innosilicon A9 ZMaster (Jun 2018) 50 kSol/s at 620 W; Z11 (2019) 135 kSol/s at 1,418 W; Z15 (2020) 420 kSol/s at 1,510 W | Z9 mini 12x, A9 29x, Z11 34x, Z15 100x (vs GTX 1080 Ti at 700 Sol/s, 250 W) | 18 | Zcash: no fork (Zcon0 vote 45 to 19 against prioritising resistance, Jun 2018; ECC chose Sapling over resistance); PoS plan announced Nov 2021. Horizen: stayed after a 51% attack (Jun 2018). Pirate: stayed | [S16] [S17] [S18] [S19] [S20] |
| 6 | Equihash parameter forks: Zhash 144,5 (Bitcoin Gold), ZelHash 125,4 (Flux), BeamHash I to III 150,5 (Beam), 210,9 (Aion), 192,7 (Zero) | Jul 2018 (BTG), Jun 2019 (Flux), Jan 2019 (Beam) | Larger memory per solver than 200,9 (Beam and Flux also changed the datapath against FPGA bitstreams) | None found for 144,5, 125,4 or 150,5. Vorick (May 2018) wrote that an Equihash chip able to follow any parameter fork had been designed | n/a | BTG 7 years, Flux 7 years, Beam 7 years, with small prizes (section 2.5) | BTG: forked after the May 2018 51% attack, attacked again Jan 2020. Beam: planned "one or two hard forks" then BeamHash III (Jun 2020) as the last. Flux: stayed on ZelHash | [S21] [S22] [S23] [S24] |
| 7 | Scrypt-N, Lyra2RE, Lyra2REv2 (Vertcoin) | Jan 2014, Dec 2014, Aug 2015 | Memory-hard sponge (Lyra2) inside a hash chain; Scrypt-N grew N over time | Dayun Zig Z1, Sep 2018, 6.8 GH/s at 1,200 W (FPGA bitstreams of about 216 MH/s per board preceded it in 2018) | Z1 20x (vs GTX 1080 Ti at 59 MH/s, 207 W) | 37 (Lyra2REv2) | Lyra2REv3, Feb 2019, "to rid the network of the current generation of ASICs and FPGAs"; 51% attacks via rented hash in Oct to Dec 2018 (22 reorgs) and Dec 2019 | [S25] [S26] [S27] [S28] |
| 8 | Verthash (Vertcoin) | Jan 2021 | A 1.2 GB file generated from the chain's own block headers; random reads; bandwidth, like Ethash | None found | n/a | 69 and counting, small prize | No fork since | [S29] [S30] |
| 9 | X16R (Ravencoin) | Jan 2018 | 16 hashes in an order set by the previous block hash; compute, with order randomised | OW Miner OW1, Sep 2019, about 182 MH/s at 1,400 W; SKC Turing R1 claimed. Widely thought FPGA-based | OW1 1.3x (vs GTX 1080 Ti at 18 to 31 MH/s, 190 to 284 W) | 20 | X16Rv2, 1 Oct 2019 (hashrate fell 70%); FPGA bitstreams for X16Rv2 within weeks (BittWare CVP-13 at 240 MH/s) | [S31] [S32] [S33] |
| 10 | KawPow (Ravencoin; Neoxa, Clore, Meowcoin, Neurai) | 6 May 2020 | ProgPoW 0.9.4 variant: random math per block, DAG, 16 KB cache reads; targets the GPU datapath | None found as of 2026 (no vendor lists one; one retailer listing naming an "Antminer X9" for KawPow is an error) | n/a | 77 and counting | Roadmap: "No additional future algorithm forks are envisaged" | [S34] [S35] [S36] |
| 11 | MTP (Zcoin, now Firo) | Dec 2018 | Argon2d memory array with a Merkle tree (Biryukov and Khovratovich, "Egalitarian computing"); 4 GB; memory size | None | n/a | 34, then replaced | Dinur and Nadler broke the 2 GB instance to under 1 MB at a 170x compute penalty before launch (2017); MTP 1.2 patched it. FiroPoW (ProgPoW variant) Oct 2021 for block size and GPU fairness, not for a chip | [S37] [S38] [S39] [S40] |
| 12 | FiroPoW (Firo) | 26 Oct 2021 | ProgPoW 0.9.4 with a per-block program | None found | n/a | 59 and counting | Nov 2025 fork cut the maximum DAG to 6.76 GB to keep 8 GB cards | [S40] [S41] |
| 13 | Blake-256 14r (Decred) | Feb 2016 | Compute; chosen to be ASIC-friendly ("easy and fast implementation of hardware is the main design goal") | Innosilicon D9, about Apr 2018, 2.4 TH/s at 1,000 W; Obelisk DCR1 (Jun 2018); Antminer DR5 (Dec 2018) 35 TH/s at 1,610 W | D9 130x, DR5 1,200x (vs GTX 1080 Ti at 4.6 GH/s, 250 W) | 26 | Embraced by design | [S42] [S43] [S44] |
| 14 | Blake2b (Sia) | Jun 2015 | Compute | Antminer A3, Jan 2018, 815 GH/s at 1,186 W; Obelisk SC1 (Jul 2018) 550 GH/s at 500 W; Innosilicon S11 (2018) 3.83 TH/s at 1,380 W | A3 58x, S11 230x (vs GTX 1080 Ti at 2.96 GH/s, 250 W) | 31 | Fork at block 179,000 (Oct 2018) to brick Bitmain and Innosilicon units and keep Obelisk's; Innosilicon then held about 37% of hashrate | [S45] [S46] [S47] |
| 15 | Blake2s (Kadena) | Nov 2019 | Compute; chosen to be "GPU mineable and not immediately ASIC mineable (but for which an ASIC can be made)" | Goldshell KD2 and KD5, Mar 2021, 18 TH/s at 2,250 W; Antminer KA3 (Sep 2022) 166 TH/s at 3,154 W | KD5 195x, KA3 1,280x (vs RTX 3080 at 8.9 GH/s, 217 W) | 16 | Embraced by design | [S48] [S49] [S50] |
| 16 | CryptoNight (Bytecoin, Monero) | Jul 2012, Apr 2014 | 2 MB scratchpad sized to a per-core L3, AES rounds, random reads; latency at SRAM scale | Secret chips from about late 2017 (85% of the hashrate vanished at the April 2018 fork); Antminer X3, announced Mar 2018, 220 kH/s at 550 W; Baikal Giant-N | X3 40x to 50x (vs Vega 64 at about 2 kH/s, 200 to 250 W, approximate) | 43 to the secret chips, 47 to the announced one | Forks: CryptoNight v7 (6 Apr 2018), v8 (18 Oct 2018), CryptoNight-R (9 Mar 2019, random math per block seeded by height, chip latency up 2.5x), RandomX (30 Nov 2019). MoneroCrusher's nonce analysis (Feb 2019) found chips at over 85% of the hashrate again, four months after v8 | [S51] [S52] [S53] [S54] [S55] |
| 17 | RandomX (Monero; Wownero, ArQmA, Zephyr, Tari) | 30 Nov 2019 | A VM running 8 chained random programs per hash on a superscalar CPU with floating point; 2 GiB dataset derived from a 256 MiB cache by a random superscalar program (SuperscalarHash); 2 MiB scratchpad in L1, L2, L3 tiers | Antminer X5, Sep 2023, 212 kH/s at 1,350 W (RISC-V cores); Antminer X9, Jul 2026 delivery, 1 MH/s at 2,472 W; Pinecone INIBOX R1X, Mar 2026, 1.2 MH/s at 2,055 W | X5 at parity with a Ryzen 9 7950X (157 against about 200 H/J, approximate); X9 and R1X 2x to 3x over the best CPU (approximate). GPUs are 25x worse per joule than CPUs on it | 46 to parity hardware, about 75 to a 2x to 3x chip | No fork as of Oct 2026. Four audits in 2019 (Trail of Bits, X41, Kudelski, QuarksLab) found nothing critical | [S56] [S57] [S58] [S59] [S60] |
| 18 | Cuckoo Cycle (Grin Cuckaroo lane, Aeternity, Cortex) | Jan 2019 (Grin) | Find a 42-cycle in a random graph; lean solver one bit per edge; memory latency, or bandwidth in the mean solver | None on the Cuckaroo lane (Cuckaroo29 tweaked every 6 months: Cuckarood Jul 2019, Cuckaroom Jan 2020, Cuckarooz Jul 2020) | n/a | 24, retired on schedule | The lane was built to die: 90% of reward at launch falling to 0% in Jan 2021 (HF4) | [S61] [S62] [S63] |
| 19 | Cuckatoo31+ (Grin's chip lane) | Jan 2019 | Same, with plain bits in place of ternary counters to simplify chips; "Proof of SRAM" per Tromp | Obelisk GRN1 announced Jan 2019 and cancelled Jul 2019; Innosilicon G32 announced 2019 and never shipped; iPollo G1, Dec 2020, 36 GPS Cuckatoo32 at 2,800 W | G1 about 4x (vs RTX 3090 at about 1 GPS, 300 W, approximate) | 23, by design | Surrender by schedule. Tromp's $10,000 linear TMTO bounty was claimed in Apr 2025 (N/k bits at about k + 1,000 hashes per edge) | [S61] [S64] [S65] [S66] |
| 20 | ProgPoW (Ethereum proposal; Bitcoin Interest, Sero, Zano as ProgPowZ, Quai) | EIP May 2018; Bitcoin Interest 2018; Quai Jan 2025 | Random math per period on a 32-register file, 16 KB cache reads, 256-byte DAG loads, keccak-f800; "saturate the GPU" | None | n/a | 8 years across its adopters | Ethereum: tentatively approved Jan 2019 and Feb 2020, petition 27 Feb 2020, left "approved" and unscheduled on 6 Mar 2020, dead. Audits: Least Authority (Sep 2019) and Bob Rao (Sep 2019). Kik's 64-bit-seed exploit (Mar 2020) patched in 0.9.4 | [S67] [S68] [S69] [S70] [S71] [S72] |
| 21 | Autolykos v1 and v2 (Ergo) | Jul 2019; v2 Feb 2021 | v1: memory-hard with a per-miner secret key, so puzzles could not be outsourced to pools. v2: the secret removed (contract pools bypassed it); a 2 GB table that grows 5% per 51,200 blocks from block 614,400 | None | n/a | 87 and counting, small prize | v2 by EIP-0009 at block 417,792 | [S73] [S74] |
| 22 | Octopus (Conflux) | Oct 2020 | Ethash-style DAG; the "dense matrix step" could not be verified in `conflux-rust` tonight (unverified) | None | n/a | 72 and counting | CIP-102 (Aug 2022) proposed switching to Ethash to attract post-Merge miners; dormant | [S75] [S76] |
| 23 | kHeavyHash (Kaspa; Bugna kept it) | Nov 2021 | cSHAKE256, a 64x64 4-bit matrix multiply from the pre-PoW hash, cSHAKE256; compute, designed for optical and specialised hardware | IceRiver KS0, Jul 2023, 100 GH/s at 65 W; KS1, KS2 (Sep 2023); Antminer KS3, Aug 2023, 8.3 TH/s at 3,188 W; KS5 Pro (Mar 2024) 21 TH/s at 3,150 W | KS0 250x, KS5 Pro 1,100x (vs RTX 3090 at 910 MH/s, 150 W) | 17 to 20 | Embraced (Sompolinsky, May 2023: "an overall positive"). Hashrate went from under 100 PH/s to over 700 PH/s in months; the GPU share was negligible by late 2023 (approximate). Forks that left: Karlsen (FishHashPlus, Sep 2024), Pyrin (PyrinHash v2, Sep 2024), Spectre (CPU AstroBWTv3), Nexellia, Waglayla, Cryptix, Hoosat | [S77] [S78] [S79] [S80] [S81] |
| 24 | NexaPow (Nexa) | 2023 | SHA-256 plus a secp256k1 Schnorr signature per attempt; framed as "useful ASICs" | DragonBall A21, Jan 2025, 3.4 GH/s at 1,800 W | 3x to 4x (vs RTX 3090 at 123 to 137 MH/s, 230 W, approximate) | 24 | None | [S82] [S83] |
| 25 | Blake3 (Alephium; Iron Fish until 2024) | Nov 2021 | Double Blake3; compute; chosen as ASIC-friendly | Goldshell AL-BOX, 2023, 360 GH/s at 180 W; IceRiver AL0; Antminer AL1 (2024) 15.6 TH/s at 3,510 W; AL3 | AL-BOX 156x, AL1 350x (vs RTX 3090 at 2.3 GH/s, 180 W) | 22 to 24 | Alephium embraced. Iron Fish forked to FishHash (Apr 2024, FIP-3: Ethash-derived, fixed 4.6 GB dataset, 512 iterations, 128-byte mix); Karlsen adopted FishHashPlus (Sep 2024) | [S84] [S85] [S86] [S87] |
| 26 | FishHash (Iron Fish, Karlsen) | Apr 2024 | Ethash-derived, 4.6 GB fixed dataset; bandwidth | None found | n/a | 30 and counting, small prize | None | [S87] |
| 27 | Eaglesong (Nervos) | Nov 2019 | Compute (a new sponge) | Toddminer C1 (Feb 2020); Antminer K5, Mar 2020, 1.13 TH/s at 1,580 W; Goldshell CK5 (Mar 2021) 12 TH/s at 2,400 W | K5 70x, CK5 500x (vs RTX 3090 at about 2.1 GH/s, approximate) | 4 | Embraced | [S88] [S89] |
| 28 | Blake2b + SHA3 (Handshake) | Feb 2020 | Compute | Goldshell HS1, Jun 2020; HS3 (Jul 2020) 2 TH/s at 2,000 W; HS5 | Over 100x (approximate) | 5 | Embraced | [S90] [S91] |
| 29 | SHA512/256d (Radiant) | 2022 | Compute | DragonBall A11; IceRiver RX0, Sep 2024, 260 GH/s at 100 W | About 550x (vs RTX 3090 at 1.3 to 1.5 GH/s, approximate) | About 24 | None | [S92] [S93] |
| 30 | ProgPowZ (Zano), DynexSolve (Dynex), Janushash (Warthog), XelisHash v1 and v2 (Xelis), VerusHash 2.2 (Verus) | 2019 to 2024 | ProgPoW variant; GPU "neuromorphic" useful work; a product of VerusHash and SHA256t to balance CPU and GPU; CPU and GPU balanced; AES-based CPU hash | None found for any of them | n/a | Small prizes throughout | Xelis forked to v2 (Jul 2024) for FPGA resistance | [S94] [S95] [S96] [S97] [S98] |
| 31 | Ethash on EthereumPoW (ETHW) after the Merge | Sep 2022 | As Ethash | The Ethash chips above | About 4x (E9 Pro, X16-P vs RTX 3090, approximate) | n/a | Embraced | [S15] |
### 1.2 What the rows say when sorted
| Class of hash | Rows | Months to first chip | First-chip gain per joule | Best gain reached |
|---|---|---|---|---|
| Compute only (chains of hashes, Blake family, SHA-3 family, matrix multiply) | 2, 13, 14, 15, 23, 25, 27, 28, 29 | 4 to 31 (median about 24) | 33x to 250x | 500x to 1,280x |
| Memory at SRAM scale (128 KB scrypt, 2 MB CryptoNight) | 1, 16 | 27, 43 | 19x, 40x to 50x | 1,100x (Scrypt, 2021) |
| Memory size without a bandwidth bound (Equihash, Lyra2REv2) | 5, 7 | 18, 37 | 12x, 20x | 100x |
| Memory bandwidth at DRAM scale (Ethash, Verthash, FishHash, Etchash) | 3, 4, 8, 26 | 32 (Ethash); none for the others | 1.1x to 1.6x | 2.9x to 4.8x |
| Random program on a commodity datapath (RandomX, ProgPoW family, X16R's order randomisation) | 9, 10, 12, 17, 20, 30 | X16R 20 (FPGA-class, 1.3x); RandomX 46 to parity; none for ProgPoW's adopters in 8 years | 1.3x (X16R), 1x (RandomX 2023) | 2x to 3x (RandomX 2026, approximate) |
| Graph search (Cuckoo) | 18, 19 | 23 on the chip lane; never on the tweaked lane | 4x | 4x |
Two caveats on the random-program rows. The prizes were small: Ravencoin, Firo and Zano never reached the market caps at which the 2018 chips appeared (section 2.5), so "no chip" is partly an economic fact. And RandomX's chips arrived once Monero's reward justified them: parity hardware at 46 months, a 2x to 3x chip at about 75 months (approximate), on a hash whose whole purpose was to make the CPU the chip.
## 2. The academic side
### 2.1 Memory-hard functions
| Paper | Result | What it means for Igneum |
|---|---|---|
| Abadi, Burrows, Manasse, Wobber, "Moderately hard, memory-bound functions", NDSS 2003 and ACM TOIT 2005 [P1]; Dwork, Goldberg, Naor, "On memory-bound functions for fighting spam", CRYPTO 2003 [P2] | The origin of the idea: CPU speed varies 100x across machines, memory latency does not, so a cost function bound by cache misses is fairer than one bound by cycles | Igneum's latency-bound rule is this argument from 2003 applied to GPUs and DRAM: the DRAM row cycle is the same physics for a chip and a card (section 2.6) |
| Percival, "Stronger key derivation via sequential memory-hard functions", BSDCan 2009 [P3] | Defines sequential memory-hardness; ROMix is sequential memory-hard in the random-oracle model; cost measured in area-time (dollar-seconds) | The area-time measure is the one the chip model uses (equal silicon); scrypt's 2011 deployment at 128 KB ignored the paper's own scale |
| Alwen and Serbinenko, "High parallel complexity graphs and memory-hard functions", STOC 2015 [P4] | Cumulative memory complexity (CMC) in the parallel random-oracle model; earlier sequential measures fail against parallel, amortising adversaries | A chip is a parallel, amortising adversary; any Igneum claim about the dataset must be made in a parallel model |
| Alwen and Blocki, "Efficiently computing data-independent memory-hard functions", CRYPTO 2016 [P5]; "Towards practical attacks on Argon2i and Balloon hashing", EuroS&P 2017 [P6] | Any data-independent MHF can be computed in less than n^2 cumulative memory; Argon2i at O(n^1.75 log n), Catena and Balloon at O(n^1.67); the attacks are practical at real parameters | Igneum's addresses are data-dependent (register state), which is the right side of this result; the price is cache-timing leakage, which does not matter for a PoW |
| Alwen, Chen, Pietrzak, Reyzin, Tessaro, "Scrypt is maximally memory-hard", EUROCRYPT 2017 [P7] | scrypt's CMC is Omega(n^2 w) in the parallel ROM, optimal, against parallel amortising adversaries | Data-dependent chains of reads are the construction with the proof; Igneum's item derivation (8 dependent cache reads) is a short chain of this kind, with no proof |
| Biryukov, Dinu, Khovratovich, "Argon2", EuroS&P 2016 [P8]; Boneh, Corrigan-Gibbs, Schechter, "Balloon hashing", ASIACRYPT 2016 [P9] | Argon2d: a one-pass adversary can cut memory at most 3x at equal area-time; Argon2i needs over 10 passes to resist the Alwen-Blocki attack. Balloon: provable in the sequential model only; the paper says parallel ASIC attacks are outside its model | A "memory-hard" label without a stated adversary model has been wrong three times in this list (Argon2i, Balloon, Catena) |
| Biryukov and Khovratovich, "Tradeoff cryptanalysis of memory-hard functions", ASIACRYPT 2015 [P10]; Forler, Lucks, Wenzel, "Catena", 2013 [P11]; Simplicio et al., "Lyra2", IEEE TC 2016 [P12] | The ranking trade-off attack on Lyra2, yescrypt and Argon2; Catena's proofs flawed, 25x area-time cut; designers changed their algorithms | Lyra2REv2 (row 7) carried this construction into a PoW and still fell to a chip at 20x; the cryptanalysis found the shortcut before the chip did |
### 2.2 Bandwidth-hard functions
| Paper | Result | What it means for Igneum |
|---|---|---|
| Ren and Devadas, "Bandwidth hard functions for ASIC resistance", TCC 2017 [P13] | Memory-hardness (CMC) bounds a chip's area advantage and says nothing about energy; energy spent on off-chip memory traffic is comparable for a chip and a CPU, so bandwidth-hardness is the lever; scrypt, Catena-BRG and Balloon are bandwidth-hard with suitable parameters; the stacked double butterfly is capacity-hard and not bandwidth-hard | The chip model's "equal silicon" row is an area argument. The energy argument is the one the Ethash chips answered: they moved the same bytes at lower energy per byte with custom memory controllers (rows 3 and 4). Igneum's hash moves 128 x 64 B = 8 KB of DRAM lines per hash on AMD and 128 x 32 B on NVIDIA; a chip with 4-byte access granularity moves 512 B for the same work. That is the bandwidth-per-watt gain the plan's last section warns about, stated in Ren-Devadas's units |
| Blocki, Ren, Zhou, "Bandwidth-hard functions: reductions and lower bounds", CCS 2018 [P14] | Bandwidth cost in the parallel ROM equals the red-blue pebbling cost of the graph; high CMC implies high bandwidth cost; Argon2i and DRSample are maximally bandwidth-hard; a tight lower bound on scrypt's energy | The right formal target for a future proof about the item derivation, if one is ever attempted; none exists today |
| Alwen, Blocki, Harsha, "Practical graphs for optimal side-channel resistant MHFs", CCS 2017 [P15] | DRSample: a practical graph with maximal depth-robustness | Not applicable: Igneum does not need side-channel resistance |
### 2.3 Asymmetric, egalitarian and graph proofs of work
| Paper | Result | What it means for Igneum |
|---|---|---|
| Biryukov and Khovratovich, "Equihash", NDSS 2016 [P16] | Wagner's generalised birthday with algorithm binding; claimed 1,000x compute for halving memory | The claim did not survive contact with a chip design: 144 MB in practice fitted the Z9's memory system (row 5) |
| Biryukov and Khovratovich, "Egalitarian computing", USENIX Security 2016 [P17]; Dinur and Nadler, "Time-memory tradeoff attacks on the MTP proof-of-work scheme", CRYPTO 2017 [P18] | MTP: Argon2d plus a Merkle tree. Dinur-Nadler: malicious proofs with under 1 MB in place of 2 GB at a 170x compute penalty, by injecting blocks that steer Argon2d's data-dependent addressing | The attacker who controls the memory's contents controls the addresses. In Igneum the day key comes from a VDF of chain state and the cache fill is a chained block function, so no miner chooses the contents. The analogy still holds for the unreviewed mixer: a structural weakness in M_r is the shortcut this paper found in MTP |
| Tromp, "Cuckoo Cycle", BITCOIN 2015 [P19]; Andersen, "A public review of Cuckoo Cycle", 31 Mar 2014, and "Exploiting time-memory tradeoffs in Cuckoo Cycle", 1 Aug 2014 [P20]; the linear TMTO bounty, claimed Apr 2025 [S66] | Edge trimming cut memory about 50x for about 2x time, two months after publication; Tromp adopted it. The 2025 bounty result: an N/k-bit chip must hash each edge about k + 1,000 times | A time-memory claim is a curve, and the curve was wrong by 50x until someone drew it. Igneum's curve between "store everything" and "recompute everything" has not been drawn (O-1.6) |
| Georghiades, Flolid, Vishwanath, "HashCore", 2019 [P21] | "Inverted benchmarking": random widgets modelled on SPEC CPU workloads so the CPU is already the chip | The same idea as RandomX and ProgPoW stated generally: the hash is a benchmark of the target hardware |
### 2.4 Program-based proofs of work and their audits
| Document | What it says | What it means for Igneum |
|---|---|---|
| RandomX `doc/design.md` and `doc/specs.md` (tevador) [S56] [S57] | A VM so that the work is "data and code"; 8 chained programs per hash so a miner cannot filter (filtering 25% of programs at a 50% speedup yields 0.44x honest speed); SuperscalarHash of about 450 instructions with 155 multiplies, scheduled for a superscalar core at a 170-cycle latency to match DRAM, so a light-mode chip with the 256 MiB cache on die pays 760 cycles and 1,240 multiplies per item, "energy comparable to loading 64 bytes from DRAM"; a 2,080 MiB dataset because a 2 GiB SRAM die was "questionable" at 7 nm in 2019; cache-to-dataset ratio capped at 8 to keep the area-time product constant; double-precision floating point to force the whole CPU; "DRAM cannot do more than about 25 million random accesses per second per bank group" | Igneum rebuilt the idea for a GPU. The parts that carried over: the dataset above on-chip cache, the dependent item derivation, the cache-to-dataset ratio (4 at genesis, 8 at year 4 under option C). The parts that did not: per-hash programs (a GPU cannot JIT per hash and stay a GPU), floating point (vendor rounding), a random item-derivation program (Igneum's mixer has a fixed shape with drawn constants). Section 4.3 ranks the last of these |
| Trail of Bits audit of RandomX, 2 Jul 2019 [S60]; Kudelski, X41, QuarksLab (2019) | Two low findings and 47 brittle parameters; the design affirmed | Four paid external reviews before launch, for a hash whose whole value was the resistance claim. Igneum has had none (ledger M7) |
| EIP-1057 ProgPoW [S67]; Least Authority audit, 9 Sep 2019 [S69]; Bob Rao hardware audit, Sep 2019 [S70] | Claimed chip gain 1.1x to 1.2x. Least Authority: no issues, five suggestions, one of them the light-evaluation attack (on-the-fly DAG generation with the 16 MB cache in on-die SRAM) "may become possible within a few years" once about 100 MB of fast on-die SRAM is feasible. Rao: energy per hash is the only meaningful metric; shipping Ethash chips show about 1.6x hashrate per watt; conventional compute chips gain little on ProgPoW; integrating the DAG on die cuts data-movement energy by over 10x, so "ProgPOW ASICs with << 0.1X E/H over GPUs can be built"; an advanced-node chip is "$20M+" and "1+ year"; a 16-die split holding a 2.78 GB DAG was about $172 per board in 2019 against about $240 for a GPU board, and a monolithic die "viable around 2025" | The light-evaluation attack is Igneum's M16 recompute chip. ProgPoW left it as a suggestion; Igneum priced it and spent the mixer against it (x8). Rao's "$172 per board for a 16-die split" is the HBM-class partial-store chip in a different form, and it is the row the Igneum model lacks |
| Kik, "ProgPoW exploit", 4 Mar 2020 [S71] | A 64-bit seed lets a chip skip memory access with a cooperating node; patched in 0.9.4 | Igneum's seed is 256 bits and the program is the epoch's; the nearest analogue is header grinding for cache locality, unmeasured (section 4.3, check 4) |
### 2.5 The economics of a chip
**What a chip costs to make** (design plus masks, by node; all figures from the cited articles, which disagree with each other by 2x and say so):
| Node | Mask set | Full design, IBS as quoted by Semiengineering (2018 and 2021) | Full design, other estimates | Sources |
|---|---|---|---|---|
| 65 nm MPW shuttle | n/a | n/a | Europractice 2025: about €51,000 minimum (9 mm^2 at €5,720 per mm^2) | [E1] |
| 28 nm | "beyond $1M" (SemiAnalysis 2022); $1M to $3M (Silicon Analysts 2026) | $51.3M (2018); $40M (2021) | $5M to $30M total NRE for a small chip (Silicon Analysts) | [E2] [E3] [E4] |
| 16/12 nm | n/a | $106M (2018 revision of a 2014 $310M figure) | Europractice MPW 16 nm: about €125,000 minimum | [E1] [E4] |
| 7 nm | "beyond $10M" (SemiAnalysis); $5M to $10M (Silicon Analysts) | $297.8M (2018); $217M "mainstream" (2021); Semiengineering's own 2023 discount: about $160M | Startups shipped 7 nm chips for "$50M to $75M" all-in (SemiAnalysis); a 10 nm-class mining chip "$20M+" (Rao 2019) | [E2] [E3] [E4] [S70] |
| 5 nm | $10M to $20M | $542.2M (2018); $416M (2021); about $280M discounted (2023) | Marvell 2023: $449M (secondary source, approximate) | [E2] [E3] [E5] |
| 3 nm | "$40M range" | $500M to $1.5B (2018); $590M (2021) | Marvell 2023: $581M (secondary, approximate) | [E2] [E3] [E5] |
Miner-makers' own numbers: Bitmain's 2018 filing shows R&D of $73M in 2017 and $86M in the first half of 2018 and three failed chips at a reported combined cost of about $500M [E6]; Canaan's 2019 prospectus shows R&D of $26.5M in 2018 and "seven tape-outs" at a 100% success rate [E7]; Vorick wrote that Bitmain brought the Sia A3 to market for "less than $10 million" and took over $20M of orders within eight minutes [S47]; Obelisk's DCR1 was a 28 nm part [S43]; Taylor's 2013 survey gives $150,000 for a 130 nm and $500,000 for a 65 nm Bitcoin chip NRE in 2012 [E8].
**Where the 256 MiB cache lands a chip.** The Counter ASIC 2.0 analysis priced the 256 MiB SRAM mirror at 128 mm^2 and $46 per good die at N5 on the shipped-product density (`sram-mirror.md` revision 2, from AMD V-Cache 64 MB on 41 mm^2 at N7 [E9], TSMC N5 HD macro 31.8 Mib/mm^2 [E10]). The history adds the node question: a cheap chip is a 28 nm chip ($1M to $3M of masks, a $5M to $30M project), and a 28 nm bit cell is about 6x an N7 cell (approximate, from memory: TSMC 28 nm HD about 0.127 um^2 against N7's 0.027 [E10]), so 256 MiB at 28 nm is about 1,000 mm^2 of SRAM on the V-Cache density: more than a reticle. The cache forces the recompute chip onto a 7 nm or better node, which moves its project from the $5M class to the $50M class (SemiAnalysis's 7 nm startup figure). That is a stronger statement than the $46 per die, and it is the reason the cache size matters more than its per-die cost. Option C (the cache doubles with the dataset) keeps it true as nodes shrink: at the 6% per year density trend the status file cites, a 512 MiB mirror in year 4 costs more mm^2 than 256 MiB today.
**When chips appeared** (CoinMarketCap historical snapshots pulled by the research agent; daily issuance is arithmetic from each chain's schedule; all approximate):
| Chain | First public chip | Market cap then | Daily issuance then (USD) |
|---|---|---|---|
| Litecoin | Gridseed, Dec 2013 | $817M | $1.0M |
| Dash | PinIdea DR-100, Aug 2017 (iBeLink 2016 widely cited, date unverified) | $2.2B | $0.6M |
| Siacoin | Obelisk SC1 announced Jun 2017; Antminer A3 Jan 2018 | $430M; $1.5B | $0.43M; $1.1M |
| Decred | Obelisk DCR1 Jun 2017; Innosilicon D9 Apr 2018 | $216M; $353M | $0.18M |
| Monero | Antminer X3, Mar 2018 (secret chips from early 2017 per Vorick, unverified) | $3.3B | $0.75M |
| Ethereum | Antminer E3, Apr 2018 | $37.4B | $7.6M |
| Zcash | Z9 mini, May 2018 | $1.1B | $2.1M |
| Bitcoin Gold | the same chips, May 2018 | $1.3B | $0.14M |
| Grin | GRN1 announced Jan 2019 (cancelled); G32 Apr 2019 (never shipped); iPollo G1 Dec 2020 | $23M (Apr 2019); $23M (Dec 2020) | $0.24M; $33K |
| Nervos | Toddminer C1, Feb 2020 | $75M | $65K to $85K |
| Handshake | Goldshell HS1, Jun 2020 | $30M | $31K |
| Kadena | Goldshell KD5, Mar 2021 | $42M | $21K |
| Kaspa | IceRiver KS0, Jul 2023 | $480M | $0.42M |
| Alephium | Goldshell AL-BOX, May 2024 | $179M | $0.1M |
| Radiant | IceRiver RX0, Sep 2024 | $18M | $22K |
Sources: [E11] (the research agent's CoinMarketCap pulls, dates in the table) and the chip rows above. Reading: a compute-bound hash gets a chip at $20K to $30K of daily issuance (Radiant, Kadena, Handshake); the 2018 cluster sat at $0.15M to $2M a day. Vorick's rule from May 2018: any coin with over $20M of block reward in a year (about $55K a day) should assume a secret chip [S47]. For Igneum the clock is the day its issuance in dollars crosses about $50K; a memory-bound hash buys time against that clock (Ethash: 32 months at the largest prize in the table), and the random program buys more (section 1.2), but nothing in the table says it buys forever.
### 2.6 Latency as the resource
| Source | What it says | What it means for Igneum |
|---|---|---|
| Li, Reddy, Jacob, "A performance and power comparison of modern high-speed DRAM architectures", MEMSYS 2018 [L1] | Row timings from datasheets: DDR4 tRCD 14, tRAS 33, tRP 14 ns; GDDR5 tRCD 14, tRAS 28, tRP 12; HBM and HBM2 tRCD 14, tRAS 34, tRP 14. Row cycle tRC about 40 ns (GDDR5) to 48 ns (DDR4, HBM2). "The memory-latency problem does still remain" | The row cycle is the floor under every dependent random read whatever the controller; HBM does not shorten it. What HBM and a custom controller change is the number of rows that can be opened per second per watt (channels, banks, pseudo-channels), which is the throughput of random reads in flight, which is what the 9070 XT probe measured as the card's ceiling (2.4 G reads/s against the 5090's 17.5 G) |
| Chang, CMU thesis, Dec 2017 [L2] | Over two decades DRAM capacity improved 128x, bandwidth 20x, latency 1.3x | The latency-bound rule has a long half-life; the bandwidth-per-watt lever (the Ethash chips) does not stand still |
| NVIDIA profiling guide and the Ampere tuning deck [L3] | L1 and L2 lines are 128 bytes in four 32-byte sectors; DRAM-to-L2 transactions default to 64 bytes since Volta, configurable 32, 64 or 128 on A100 | The 5090's 32-byte sector per 4-byte read is where a custom controller gains bandwidth efficiency (8x fewer bytes), the Ren-Devadas energy lever; the 9070 XT's 64-byte line is 16x. Neither changes the row cycle |
| RandomX `doc/design.md` [S56] | About 25 million random accesses per second per DRAM bank group; "all Dataset accesses read one CPU cache line (64 bytes) and are fully prefetched"; one program iteration tuned to "typical DRAM access latency (50-100 ns)" | The same arithmetic Igneum uses (128 dependent reads per hash against the card's random-read ceiling), from the design that held longest |
| Condrey, "PoSME", arXiv Apr 2026 (single author, not peer reviewed) [L4] | Latency-bound pointer chasing with hash compute under 3.5% of the step cost; GPUs 14x to 19x slower than a consumer CPU | The only dedicated latency-bound PoW paper found; its GPU-vs-CPU gap is the cost RandomX pays, and the cost Igneum avoids by keeping thousands of loads in flight per card |
The paper that does not exist: nothing found treats cache timing as a feature; Catena and Argon2i treat it as a leak. No peer-reviewed survey of ASIC resistance as such surfaced; the nearest are Cho's 2018 multi-hash evaluation [P22] (X11-style resistance "is not strong enough"), Feng and Luo's 2020 three-processor study [P23] (GPUs dominate CryptoNight, Ethash and Cuckoo on CPU, GPU and Xeon Phi) and Yaish and Zohar's 2023 pricing of mining hardware as a bundle of options [P24].
## 3. The lessons
Each lesson is stated once, with the rows it comes from.
1. **Compute-bound work loses by 30x to 1,000x within two years, whatever its shape.** Chains of eleven hashes (row 2), sixteen hashes in a random order (row 9), a 64x64 matrix multiply (row 23), a new sponge (row 27), a signature per attempt (row 24): every one got a chip, the random-order chain at 1.3x by an FPGA within 20 months and the rest at 33x to 1,100x. Multiplying the number of fixed functions multiplies the chip's die, not its difficulty. For Igneum: nothing in the program's ALU work is a defence and the design already says so (ledger M1); the defence is the memory path.
2. **Memory at SRAM scale is compute-bound with extra steps.** Scrypt's 128 KB (row 1) and CryptoNight's 2 MB (row 16) were sized to a 2011 and a 2014 CPU cache; a chip put the same memory on die and won 19x and 40x. Igneum's answer is the 256 MiB cache growing with the dataset (option C) and the 1 GiB to 2 GiB dataset; section 2.5 shows the cache size also sets the chip's node and therefore its project cost. The hot table (layer 5) was a step back toward SRAM scale, and the measurement agreed (the honest card paid 7% to 16%, the chip paid $0.23 per MB).
3. **Bandwidth-bound work gets a memory chip at 2x to 5x.** Ethash held 32 months and then got chips whose whole design was the memory system: DDR3 (E3, no gain), GDDR6 (A10 Pro, 1.2x), custom controllers (Linzhi, 2.1x), on-package memory (Jasminer X4, 4.8x) (rows 3, 4). Rao's audit explains why in energy terms: the chip moves the same bytes at lower energy per byte, and a split-die design holding the DAG was already cheaper than a GPU board in 2019. Ren and Devadas give the bound: a chip's energy advantage on a bandwidth-hard function is the ratio of its memory energy per bit to the GPU's. Igneum's rule "avoid leaning on bandwidth" is right; its model has no row for this chip (section 4.3, addition 1).
4. **Latency-bound and random-program work held longest, and the prize was usually small.** CryptoNight's latency bound at SRAM scale held 43 months, then fell to secret chips (row 16). RandomX's at DRAM scale held 46 months to parity hardware and about 75 to a 2x to 3x chip (row 17, approximate), on the largest prize any resistant hash has carried. ProgPoW's adopters have had no chip in 8 years on small prizes (rows 10, 12, 20). Verthash, Autolykos, Octopus and FishHash have none on small prizes (rows 8, 21, 22, 26). The honest reading: the random program on a DRAM-latency bound is the strongest construction the history has, and nobody has tested it at Ethereum's prize.
5. **Periodic human forks fail as a defence.** Monero: four forks in 20 months; chips were back at 85% of the hashrate within four months of the v8 fork (row 16), and Vorick wrote that a chip able to survive forks at under a 5x hit had been designed. Vertcoin: three forks, two followed by rented-hash 51% attacks within weeks, because each fork reset the hashrate to a rentable size (row 7). Ravencoin: FPGA bitstreams for the new order within weeks (row 9). Sia: the fork bricked competitors' chips and left one vendor at 37% (row 14). Grin: the tweaks worked because the lane was scheduled to die (row 18). The chip's design cycle is 5 months for Bitmain (Vorick) and 13 for a startup; a fork every 6 months is a race the chip wins on the second lap, and each fork is a governance event. Igneum's draws are automatic and scheduled at genesis; that is the right side of this lesson, and section 4.3 asks whether the epoch can also be shorter than a bitstream (addition 5).
6. **RandomX got the target right and the derivation right, and it costs GPUs 25x.** Right: the work is "code and data" so a fixed circuit cannot serve it; chained programs defeat filtering (0.44x); the dataset is above any SRAM die; the item derivation is a random superscalar program tuned to DRAM latency so the light-mode chip pays as much energy per item as a DRAM read (section 2.4). The cost: a GPU runs the VM at 25x worse per joule than a CPU (row 17), which is the cost Igneum refuses, and the reason the program is per hour and compiled. What Igneum did not take: the random item derivation (addition 2) and four external audits before launch (addition 3).
7. **ProgPoW got the datapath right and lost on governance and one unpriced attack.** Right: target the commodity hardware's whole datapath (random math, register file, cache reads, DAG loads) so a chip has to be a GPU; the hardware audit agreed for compute-only chips (1.1x to 1.2x, Rao). Unpriced: the DAG on die (Least Authority suggestion 2, Rao's "<< 0.1x"), the same attack Igneum calls M16. Not adopted: two tentative approvals, a petition, bugs found late (Kik), authorship disputes and a PoS roadmap; the change needed a contentious fork on a live chain (row 20). Igneum's lesson is the one it already follows: every layer goes in before the public testnet as a genesis rule or a reserve, so no adoption vote is ever needed.
8. **What a "GPU-friendly" chain lost when its GPU miner fell behind: the miners, then the chain's shape.** Kaspa's hashrate rose 7x in months and its GPU share went to nothing; seven forks left to re-resist (row 23). Alephium, Nervos, Handshake, Kadena and Radiant went the same way without the forks (rows 25, 27, 28, 15, 29). Iron Fish forked away from its own Blake3 within a year of the first box (row 25). The chains kept their security budget and lost the fleet that had launched them; the fleet's hardware went to the next GPU chain. For Igneum the metric is the share of hashrate on consumer cards by model, which is what the January 2027 benchmark should report and what the observer can estimate earlier (addition 4).
9. **An unreviewed memory-hard construction has a shortcut until someone looks.** MTP fell from 2 GB to under 1 MB before launch (row 11); Catena's proofs were flawed; Argon2i's parameters were attackable at the IRTF's "paranoid" setting; Cuckoo's memory claim was off by 50x within two months (section 2.3). Igneum's M_r and chained cache have had no cryptanalysis (`MEMHARD.md` section 3, ledger M7); the acceptance rule is a statistical filter, not a proof. The x8 decision multiplies the mixer's weight in the chip model, which multiplies the cost of a structural weakness in it (addition 3).
10. **The secret chip is found by its share, and it is on the chain before the announcement.** Monero's chips held 85% before anyone saw them; a nonce-pattern analysis found them (row 16). Zcash's Z9 was "5x to 10x below" what Obelisk's own study said the hash allowed, which is Vorick's evidence that better secret chips existed (section 1.1 row 5). A detector costs an observer query; a bounty costs escrow (addition 4).
## 4. The audit of Igneum against the history
### 4.1 The first-generation layers (live in class v2 and carried into v3)
| Layer | Answers which failure | Does not answer | Evidence |
|---|---|---|---|
| Random program per epoch from a VDF seed, 12 integer families, nonce-dependent select | Fixed-function chips (lesson 1); program filtering and seed grinding (RandomX's 0.44x, spec 04's 130-to-1) | A "GPU without graphics": a programmable sequencer over 12 ops and 8 registers (ledger M1); an FPGA overlay or bitstream compiled within the hour (rows 7, 9: FPGAs were the first adversary of Lyra2REv2 and X16R); the 7.5x AMD gap is a one-vendor fleet | Rows 9, 10, 16, 17, 20 |
| Weak-program acceptance, exact 16 loads, fresh-source rule | Per-program hash-rate spread (1.10x residual) that a chip could pick | Nothing it claims to; a chip's advantage cannot come from the program (status 20:16) | Census [I1] |
| 1 GiB to 2 GiB dataset of 4-byte random reads, latency-bound, cache 256 MiB | SRAM-scale memory (lesson 2); the bandwidth lever at the honest card (lesson 3: 128 x 4 B keeps the 5090 at 9% of its stream bandwidth) | The partial-store chip with a custom memory system (lesson 3); the time-memory curve (O-1.6) | Rows 1, 3, 16; [P13] |
| 8 dependent cache reads per item, fixed-shape mixer with drawn constants | The on-die recompute chip (Least Authority's light-evaluation attack), priced at 2.45x bare under v2 | The fixed shape gives the chip its 3x factor (lesson 6); no cryptanalysis (lesson 9) | [S69] [S70]; M16 |
| Era draws from chain state (op weights, fold rotations), reserve families by height, dataset growth, no human release | Fork fatigue and fork-reset attacks (lesson 5); the chip that "survives forks at under 5x" (Vorick) is the chip that the draws are meant to outlast | The draws touch the program, not the item derivation, so they cost the recompute chip nothing (`chip-model-v3.md` section 2) | Rows 7, 16, 18 |
| Warp-unit CPU verification without the dataset (2.1 ms per warp under x8) | Keeps the verifier light, the Equihash and Cuckoo goal | Caps every lever: the mixer budget stops at the 10 ms gate | [P16] [P19] |
### 4.2 The Counter ASIC 2.0 layers as decided tonight
| Layer | Decision (status file) | Answers | Does not answer | History's verdict |
|---|---|---|---|---|
| 1 Load width 4, 16, 64 B | Keep 4 B (w16 closes nothing) | Keeps the 5090 latency-bound (9% of stream) | The AMD 7.5x gap (2.4 G reads/s at every width) | Right by lesson 3; the vendor gap is a 3.0 question and a soft form of lesson 8 |
| 2 Per-program width mix | Out (spread over 5% on every card) | n/a | n/a | Right: a per-program spread is what a chip picks (lesson 1's X16R: randomised order gave 1.3x, the shape still fixed) |
| 3 Per-warp scratch with RMW | Out (does not move the recompute chip; 2.4x at every share; costs GPUs 12% to 48%) | n/a | n/a | Right: SRAM-tier work favours the chip (lesson 2; Rao: SRAM is the chip's weapon) |
| 4 + 8 Era layout (stride, interleave) and per-site windows | In (era inside the class) | A hard-wired layout tuned to one era | A programmable address decoder (era-layout.md section 8 says so); costs the recompute chip nothing | Small by itself; its value is in lesson 5 (automatic change without a fork) |
| 5 Hot table sized to GPU cache | Measured, not adopted (honest card pays g = 0.84 to 0.93; chip pays SRAM) | n/a | n/a | Right by lesson 2 |
| 6 Cache growth | Option C: doubles with the dataset (256 MiB, 512 MiB year 4, 1 GiB year 12) | Keeps the mirror on a leading node (section 2.5) | n/a | Right; RandomX's cache-to-dataset ratio of 8 is reached at year 4 |
| 7 INT8 matrix family | Reserve R1 = mm8, W_new 4, unlock era 4 or 90% signal | A family that a 12-op chip lacks | Matrix hardware is the most abundant custom silicon on earth; Least Authority's suggestion 5 was "watch ML hardware"; Apple's emulation costs 1.6x to 4.7x per op | Keep in reserve, order it last (addition 6) |
| 9 Epoch length as an era parameter (10 min to 2 h) | Reserve only, design on `ca2-epoch` | The bitstream-per-epoch FPGA (rows 7, 9) | The FPGA overlay (a soft GPU) and the HBM FPGA | Rank it up (addition 5) |
| Mixer x8 (M16's lever) | In: 0.31x bare, 0.92x with the 3x factor, verifier 2.1 ms per warp, daily build 23 to 77 ms | The on-die recompute chip (lesson 6, Least Authority's attack) | Its own fixed shape (the 3x factor stays) and its lack of review (lesson 9) | The right lever; additions 2 and 3 are what the history says to do to it next |
### 4.3 Ranked additions and upgrades
Ranked by how much the history says each would change the outcome, with the cost to GPUs and the risk. "Genesis" means a rule fixed before the public testnet; "reserve" means a named family or parameter in the genesis reserve, unlockable by height or 90% signal; "nowhere" means do not add.
| Rank | Addition | What it does | Evidence | Cost to GPUs | Risk | Where |
|---|---|---|---|---|---|---|
| 1 | **Price the partial-store chip and draw the time-memory curve.** A chip that stores a fraction f of the dataset in HBM or on many narrow DRAM channels, recomputes the rest from a 256 MiB on-die cache under x8, and reads with 4-byte granularity. Rows for f = 0.25, 0.5, 1 at HBM3 and at GDDR7 random-read rates, priced in energy per hash (Rao's metric) and in reads in flight per watt | The only chip class that beat a memory-bound GPU hash: Ethash's 2.1x to 4.8x came from the memory system with no on-die dataset (rows 3, 4); Rao priced a 16-die DAG holder under a GPU board in 2019; Cuckoo's curve was wrong by 50x until drawn [P20]; O-1.6 is open and `MEMHARD.md` section 3 item 2 says the curve was never drawn | None (analysis) | The row may come out over 2x, which would qualify the public claim before anyone else does | Genesis (before the vectors freeze) |
| 2 | **A random item-derivation program per day** in place of the fixed-shape mixer: a SuperscalarHash-style generator, integer only, drawn from the day key, with its own acceptance test, compiled once a day by miners and verifiers | RandomX's reason for SuperscalarHash: a fixed derivation is hard-wired by a chip; a random one makes the light-mode chip a CPU (section 2.4). In Igneum's model the fixed shape is the 3x factor that turns 0.31x into 0.92x; removing the factor is worth more than x8 to x16 would be (x16: 0.46x with the factor by M16's table) | None per hash (the daily build is 23 to 77 ms at x8 and would roughly double); the verifier needs a per-day compiled derivation (a JIT, or a round schedule drawn from a fixed set of reviewed rounds), measured against the 10 ms gate | Cryptanalysis of random ARX programs; weak draws; a JIT in the verifier is new attack surface; the vendors must agree bit-exactly on a program they compile | Reserve (named family, unlock by height or signal) now; genesis if the verifier cost is measured under the gate before the freeze |
| 3 | **External cryptanalysis of M_r, the chained cache and the acceptance rule before genesis**, with the x8 shape as the target | Lesson 9 (MTP, Catena, Argon2i, Cuckoo); RandomX bought four audits for $141,000 before launch [S60]; the x8 decision multiplies the mixer's weight in the chip model, so a shortcut inside the mixer is now worth 8x more to a chip | None | Finding something late moves the vectors; not finding it in time moves nothing | Genesis gate (ledger M7, raised in priority) |
| 4 | **The clock and the detector.** (a) A share-pattern detector on the observer: per-program hash-rate spread, nonce-group patterns and per-card-model rate bands, with an alert when a population behaves like one fixed design (MoneroCrusher's method); (b) a stated trigger: the bounty escrowed and the benchmark live before daily issuance crosses about $50K (Vorick's rule), not on a calendar date | Lesson 10 (85% secret share); section 2.5's table (chips at $20K to $30K a day on compute-bound hashes); D11 (the bounty is unfunded) | None | A detector with false positives; a trigger the project lead has to fund | Not a layer; genesis-independent; do it before the public testnet |
| 5 | **Rank layer 9 (the epoch length) up, and measure the FPGA lane**: the compile-ahead cost per card at a 10-minute epoch (the `ca2-epoch` work), plus an estimate of a soft-overlay FPGA miner with HBM (reads in flight per watt against the 5090's 17.5 G/s) | FPGAs were the first adversary of Lyra2REv2 and X16R and came back within weeks of X16Rv2 (rows 7, 9); Xelis forked for FPGA resistance (row 30); a per-hour program is a bitstream target in a way a per-hash program is not | At 10-minute epochs: 6x the compile work per card (measured on `ca2-epoch`); the VDF lead shrinks | A short epoch moves the difficulty window (spec 1.12) and the seed path | Reserve (as decided), with the measurement before the public testnet |
| 6 | **Order the reserve by chip-unfriendliness**: families that force a full 32-bit datapath per lane first (byte permute, bit-field extract, variable shifts, popcount, select, the second shuffle form), mm8 last | Least Authority's "watch ML hardware"; int8 matrix blocks are licensable IP at every node; Apple pays 1.6x to 4.7x per emulated dot4 (status 20:38) | None at launch | None | Reserve ordering, genesis |
| 7 | **A vendor-share metric and a 3.0 target for the AMD gap**: the share of hashrate by vendor published with the benchmark, and the line-width question kept open as the plan says | Lesson 8: a one-vendor fleet is a softer version of chip capture; Equihash's NVIDIA tilt and Ethash's balance were part of each chain's miner politics (rows 3, 5) | n/a | A width that closes the gap makes the 5090 bandwidth-bound (status 20:27) | Counter ASIC 3.0 |
Checks the history suggests that are not layers:
| Check | Why | Source |
|---|---|---|
| 1. Header grinding for cache locality: can a miner search the pre-PoW header hash H for 32-lane groups whose 128 loads cluster into fewer DRAM rows or cache lines, at a search cost below the gain? | Kik's ProgPoW exploit and Dinur-Nadler's MTP attack were both "the attacker steers the addresses" | [S71] [P18] |
| 2. The chip detector's baseline: the per-program spread per card model, from the first week of the public testnet | Needed before addition 4(a) can alert | [S54] |
| 3. The 28 nm SRAM density figure in section 2.5 (approximate, from memory) and the node-cost consequence, cited properly | It is the argument that the cache size sets the chip's project cost | [E10] |
Evaluated and placed nowhere, with the reason:
| Candidate | Verdict | Reason |
|---|---|---|
| Program entropy per hash (RandomX) instead of per hour | Nowhere | Per-hash programs need an interpreter or JIT on the GPU, which is the 25x GPU penalty RandomX pays (row 17) and the reason Igneum compiles per epoch. The filtering attack per-hash chaining prevents is already closed by the VDF seed and the acceptance rule. Per-hour's residual exposure is the FPGA lane, which addition 5 addresses with a shorter epoch, not with per-hash programs |
| Superscalar dependency-chain design for the program itself | Nowhere, beyond what exists | The program's ALU work is not the defence (lesson 1); the dependency chain that matters is the 8 dependent cache reads per item and the 128 dependent loads per hash, both in place. The superscalar idea belongs in the item derivation (addition 2) |
| Verthash's table from the blockchain; a dataset derived from chain history | Nowhere | Against the recompute chip and the partial-store chip it changes nothing: both build the table from the same public inputs the GPU does. The day key already comes from a VDF of chain state, which gives the unpredictability without a history dependency; a history dependency costs the verifier the history (Verthash needs the headers) and ties the hash to pruning (spec 10) |
| Grin's dual PoW with a shifting split | Nowhere | It is a scheduled surrender (row 19). Igneum's automatic schedules (dataset growth, reserve unlocks, cache doubling) are the shifting split applied to one hash; a second lane would hand a chip a lane |
| Autolykos v1's non-outsourceability | Nowhere | It stops pools, not chips, and Ergo removed it after 19 months because contract pools bypassed it (row 21); Igneum needs pools (spec 09) |
| A per-hash VRF against nonce grinding | Nowhere | A signature per attempt is what NexaPow did and it got a 3x to 4x chip (row 24): EC arithmetic is fixed-function work. The grinding Igneum must guard is the header-locality search (check 1), which a VRF does not touch |
| Ternary or variable-precision integer ops | Nowhere, beyond the reserve | Every family must be bit-exact on three vendors; dot4 is native on NVIDIA and AMD and emulated on Apple at 1.6x to 4.7x (status 20:38), so each precision added is paid by the weakest vendor. The reserve already holds the integer-exact candidates; adding more does not change lesson 1 |
| Cache-timing-bound reads (ProgPoW's 16 KB cache, RandomX's L1 tier) | Nowhere | Measured out tonight at the L2 tier (layer 5) and the per-warp tier (layer 3): the honest card pays and the chip buys SRAM at $0.23 per MB. Rao's audit says the same about ProgPoW's cache reads |
| Divergent data-dependent branches | Nowhere (already excluded) | Branches cost a GPU divergence and a chip nothing; RandomX's single predictable branch targets speculative CPUs, which Igneum does not have |
| Floating point | Nowhere (already excluded) | Vendor rounding splits the chain (spec 1.14); RandomX could afford it because its target is one ISA family with IEEE semantics |
## 5. Decisions this raises for the project lead
| # | Decision | Recommendation |
|---|---|---|
| 1 | Add the partial-store chip rows to `chip-model-v3.md` and draw the time-memory curve before the public testnet | Yes, before the vectors freeze (addition 1) |
| 2 | Name a random item-derivation program as a reserve family, and fund the verifier measurement that would move it to genesis | Reserve now; genesis if the verifier lands under the gate (addition 2) |
| 3 | Commission the external cryptanalysis of M_r and the chained cache before genesis, with the x8 shape as the target | Yes (addition 3; ledger M7) |
| 4 | Escrow the bounty and set its trigger to daily issuance, not to a date; build the share-pattern detector on the observer | Yes to the detector now; the escrow is the project lead's (D11) |
| 5 | Rank the epoch-length reserve above the mm8 reserve, and measure the FPGA lane | Yes (additions 5 and 6) |
## 6. Sources and limits of this research
Research was gathered by four sub-agents between 20:10 and 20:45 UTC on 5 October 2026 and checked against the citations below. Fetch failures they reported: eprint.iacr.org PDFs sit behind a challenge page (abstract pages worked), so the Ren-Devadas energy figures, the Alwen-Blocki EuroS&P tables and the Lyra2 exponent come from abstracts; medium.com and bitcointalk.org returned 403 (Vorick's post was read through archive.sia.tech and secondary coverage; the IfDefElse posts through the Veil interview); Bitmain's prospectus PDF was blocked; the Dash iBeLink date and Octopus's "matrix step" are unverified; the 28 nm SRAM bit cell is from memory. Every hash-per-joule gain is derived from the cited rate and watt figures and is approximate.
Igneum sources: [I1] `docs/analysis/weak-program-census-2026-10-03.md`; `docs/plans/counter-asic-2.md` (be4b295); `docs/plans/counter-asic-2-status.md` and `docs/plans/counter-asic-2-rollout.md` (`ca2-coord`); `docs/analysis/chip-model-v3.md` (`ca2-mixer` 1ab8b21); `docs/analysis/m16-recompute-attacker-2026-10-05.md`; `docs/analysis/sram-mirror.md` revision 2 (`ca2-analysis`); `docs/plans/era-layout.md` (`ca2-era`); `docs/plans/hot-table.md` (`ca2-cache`); `docs/plans/read-width.md` (`readwidth`); `docs/spec/01-lottery-hash.md`, `04-seeds-and-vdf.md`; `docs/bench-log.md` ("the 9070 XT on the eGPU", `opencl-rdna4`); `proto-metal/MEMHARD.md`; `docs/fud-ledger.md` M1, M3, M7, M16, C2, D11.
History rows:
- [S1] https://medium.com/@Linzhi/what-is-memory-hard-45a363b59dfe (Tenebrix's 2011 claim); https://en.wikipedia.org/wiki/Litecoin
- [S2] https://jamesachambers.com/early-bitcoin-asic-miner-pictures-history/ (Gridseed); https://www.mikewesson.com/2013/04/29/mining-litecoin-on-ati-radeon-7970s/ (7970 at 700 kH/s)
- [S3] https://www.design-reuse.com/news/34403/innosilicon-28nm-litecoin-asic-reference-miner.html (A2, Apr 2014); https://www.coindesk.com/markets/2014/05/14/kncminer-reveals-additional-titan-scrypt-asic-specs (Titan)
- [S4] https://www.asicminervalue.com/miners/bitmain/antminer-l3-504mh ; https://cryptoage.com/en/2550-bitmain-antminer-l7-is-a-new-asic-miner-for-litecoin-and-dogecoin.html ; https://www.coindesk.com/markets/2014/09/11/dogecoin-community-celebrates-as-merge-mining-with-litecoin-begins
- [S5] https://docs.dash.org/en/stable/docs/user/introduction/features.html ; https://www.dash.org/news/happy-birthday-darkcoin/
- [S6] https://cryptomining-blog.com/7117-the-first-x11-mining-asic-ibelink-dm384m-asic-dash-miner/ ; https://cryptomining-blog.com/7493-power-usage-and-noise-of-the-ibelink-dm384m-x11-asic-miner/ ; https://bitcointalk.org/index.php?topic=854257.320 (R9 280X at 4 MH/s)
- [S7] https://99bitcoins.com/guides-and-tutorials/dash-mining/antminer-d3-review/ ; https://www.cryptocompare.com/mining/asic-miner-market/baikal-giant-x10-x11-10ghs/
- [S8] https://ethereum.org/developers/docs/consensus-mechanisms/pow/mining/mining-algorithms/dagger-hashimoto/
- [S9] https://cryptoslate.com/bitmain-e3-asic-ethereum-miner/ (4 Apr 2018: E3 4.44 W/MH against a tuned 1080 Ti and RX 570); https://hothardware.com/news/bitmain-launches-ethereum-asic-miner-hashrate-comparable-8-gtx-1080-gpus
- [S10] https://innosilicon.global/product/innosilicon-a10-pro-6gb-ethereum-miner-500-mh-s/ ; https://www.notebookcheck.net/The-NVIDIA-GeForce-RTX-3080-is-an-Ethereum-mining-monster-overclocked-cards-deliver-nearly-100-MH-s-double-the-Radeon-RX-5700-XT.494246.0.html
- [S11] https://www.coindesk.com/tech/2020/12/21/linzhi-begins-rollout-of-long-awaited-ethereum-miner-phoenix ; https://www.theblock.co/post/88622/questions-new-ethash-asic-ethereum (F2Pool: 2,733 MH/s at about 3,000 W)
- [S12] https://miningnow.com/asic-miner/jasminer-x4-2500mh-s/ ; https://www.asicminervalue.com/miners/bitmain/antminer-e9-2-4gh ; https://2miners.com/blog/asic-miners-for-ethereum-antminer-e3-vs-innosilicon-a10-eth-master-comparison/ (the 3% estimate)
- [S13] https://eips.ethereum.org/EIPS/eip-1057 ; https://www.theblock.co/news/ecosystems/2020-02-26-ethereum-community-members-submit-dissenting-progpow-petition-57061 ; https://ethereum.org/roadmap/merge/ ; https://cointelegraph.com/news/bitmains-antminer-e3-to-continue-mining-ether-with-new-update (the E3's 4 GB limit)
- [S14] https://ethereumclassic.org/blog/2020-11-27-thanos-hard-fork-upgrade/
- [S15] https://www.asicminervalue.com/miners/bitmain/antminer-e9-pro-3-68gh ; https://pool.kryptex.com/device/asic/jasminer/x16-p ; https://whattomine.com/coins/151-eth-ethash/asics
- [S16] https://eprint.iacr.org/2015/946 (Equihash); https://en.wikipedia.org/wiki/Zcash
- [S17] https://variance.hu/2017/05/08/748-solsec-zcash-equihash-teljesitmeny-egy-gtx-1080-ti-kartyabol/ (1080 Ti at 748 Sol/s)
- [S18] https://www.coindesk.com/markets/2018/05/03/bitmains-latest-crypto-asic-can-mine-zcash ; https://coinguides.org/innosilicon-a9-zmaster-50k-sols-equihash-asic/ ; https://support.bitmain.com/hc/en-us/articles/360012223994-Z9-Specifications ; https://www.asicminervalue.com/miners/bitmain/antminer-z11 ; https://d-central.tech/miners/antminer-z15/
- [S19] https://github.com/ZcashFoundation/zfnd/blob/master/_posts/blog/2018-05-08-statement-on-asics.md ; https://www.coindesk.com/tech/2018/06/28/zcash-votes-against-asic-resistance-in-boon-for-big-miners ; https://electriccoin.co/blog/ecc-roadmap-calls-for-focus-on-wallet-proof-of-stake-and-interoperability/
- [S20] https://blog.horizen.io/zencash-statement-on-double-spend-attack/ ; https://blog.horizen.io/horizen-zen-statement-on-mining-algorithm/ ; https://forum.zcashcommunity.com/t/list-of-all-coins-projects-on-equihash-asic-resistant-not-resistant/29085
- [S21] https://en.wikipedia.org/wiki/Bitcoin_Gold ; https://gist.github.com/metalicjames/71321570a105940529e709651d0a9765
- [S22] https://fluxofficial.medium.com/zels-custom-pow-algorithm-zelhash-activation-in-mid-june-ad3d14d72135 ; https://uploads-ssl.webflow.com/60c73eaed3399e074029d643/60fd7d883200fcde5ceb7049_ZelHash_v1.0.pdf
- [S23] https://github.com/BeamMW/beam/wiki/BEAM-Mining ; https://docs.beam.mw/BeamHashII.pdf ; https://medium.com/minerstat/beamhashiii-beam-forks-to-a-new-algorithm-at-block-777777-dd2aeacc9e5
- [S24] https://aion.theoan.com/blog/aion-mainnet-launch-kilimanjaro/ ; https://miningpoolstats.stream/zero
- [S25] https://vertcoin.io/history/ ; https://www.newsbtc.com/2014/12/01/vertcoin-introduces-new-pow-algorithm-promises-asic-free-features/
- [S26] https://cryptoage.com/en/1231-first-asic-miner-lyra2rev2-dayun-zig-z1.html ; https://www.asicminervalue.com/miners/dayun/zig-z1 ; https://whattomine.com/gpus/36-nvidia-geforce-gtx-1080-ti ; https://arxiv.org/pdf/1905.08792 (FPGA bitstreams)
- [S27] https://cryptobriefing.com/vertcoin-vtc-51-percent-attack/ ; https://en.wikipedia.org/wiki/Vertcoin ; https://www.fxstreet.com/cryptocurrencies/news/vertcoin-cryptocurrency-network-fell-victim-to-attack-51-201912030704
- [S28] https://github.com/vertcoin-project/vertcoin-core/releases/tag/0.14.0 (Lyra2REv3)
- [S29] https://soundcloud.com/vertcoin-talk/vertcoin-talk-episode-24-verthash-fork-happens-january-30th-2021 ; https://crazy-mining.org/en/software/wallets/vertcoin-vtc-instructions-for-mining-on-verthash/
- [S30] https://coincub.com/mining/how-to-mine-vertcoin-vtc/
- [S31] https://ravencoin.org/assets/documents/X16R-Whitepaper.pdf ; https://tronblack.medium.com/ravencoin-asic-thoughts-e6c0079609e6
- [S32] https://cryptoage.com/en/1782-asics-ow-miner-ow1-and-skc-miner-turing-r1-for-the-x16r-algorithm-exist.html ; https://en.cryptonomist.ch/2019/09/17/mining-ravencoin-hashrate/
- [S33] https://en.cryptonomist.ch/2019/10/02/ravencoin-rvn-hard-fork/ ; https://cryptomining-blog.com/11320-ravencoin-rvn-getting-fpga-mining-support-for-the-x16rv2-algorithm/
- [S34] https://medium.com/minerstat/kawpow-ravencoin-forks-to-a-new-algorithm-2e730cd09fb3 ; https://tronblack.medium.com/ravencoin-kawpow-expectations-a6a063df58f2
- [S35] https://github.com/RavenProject/Ravencoin/blob/master/roadmap/README.md ; https://whattomine.com/coins/234-rvn-kawpow/gpus ; https://miningreturns.com/learn/ravencoin-mining-guide
- [S36] https://www.neoxa.net/whitepaper/ ; https://woolypooly.com/en/blog/ravencoin-algorithm
- [S37] https://arxiv.org/pdf/1606.03588 (Egalitarian computing, MTP)
- [S38] https://eprint.iacr.org/2017/497 (Dinur and Nadler); http://blog.zorinaq.com/attacks-on-mtp/ ; https://firo.org/2017/07/21/mtp-audit-and-implementation-bounty.html
- [S39] https://firo.org/2018/12/05/mtp-faq-all-you-need-to-know.html
- [S40] https://firo.org/2021/10/01/firopow-and-instantsend-release.html
- [S41] https://firo.org/2025/11/19/hardfork-successful-nov-2025.html
- [S42] https://docs.decred.org/research/blake-256-hash-function/
- [S43] https://www.asicminervalue.com/miners/innosilicon/d9-decredmaster ; https://www.asicminervalue.com/miners/obelisk/dcr1 ; https://crypto.news/hardware-companies-are-launching-dedicated-asic-miners-for-decred/ (DCR1 at 28 nm)
- [S44] https://cryptoage.com/en/1254-bitmain-antminer-dr3-7,8-th-s-on-the-algorithm-blake-14r-decred.html ; https://medium.com/luxor/bitmain-antminer-dr5-decred-setup-guide-1c417f5f61fc
- [S45] https://1stminingrig.com/antminer-a3-review-bitmain-surprises-everyone-with-this-new-siacoin-miner/ ; https://medium.com/obelisk-blog/obelisk-update-may-june-2018-260fce12a825 ; https://www.eastshoremining.com/tutorial-innosilicon-s11-siamaster-3-83th-siacoin-miner/
- [S46] https://www.coindesk.com/markets/2018/10/19/sia-network-releases-hard-fork-code-to-block-crypto-mining-giants ; https://siasetup.info/learn/forks
- [S47] Vorick, "The state of cryptocurrency mining", 13 May 2018: https://archive.sia.tech/the-state-of-cryptocurrency-mining-538004a37f9b (read through https://davidgerard.co.uk/blockchain/2018/05/14/from-sia-an-incendiary-post-on-the-state-of-cryptocurrency-mining-in-2018/ and https://zycrypto.com/asic-manufacturer-shares-important-information-for-token-creators-and-miners/); Bitmain's reply https://blog.bitmain.com/en/bitmain-sia-state-cryptocurrency-mining/
- [S48] https://medium.com/kadena-io/kadena-public-blockchain-releases-fully-public-testnet-v3-hashing-algorithm-and-mining-api-e230a51c7b26 ; https://www.coindesk.com/markets/2019/11/04/kadena-goes-live-announces-new-token-sale-aiming-for-20-million
- [S49] https://www.asicminervalue.com/miners/goldshell/kd5 ; https://asicmarketplace.com/product/goldshell-kd2-kadena-miner-6-4-th-s/ ; https://asicmarketplace.com/product/bitmain-antminer-ka3-kadena-miner-166th/
- [S50] https://minerstat.com/hardware/nvidia-rtx-3080-lhr
- [S51] https://bytecoin.org/old/whitepaper.pdf ; https://docs.getmonero.org/proof-of-work/cryptonight/ ; https://en.wikipedia.org/wiki/CryptoNote
- [S52] https://news.8btc.com/bitmain-to-release-antminer-x3-cryptonight-asic-miner-with-220-khs-hashrate ; https://bitcointalk.org/index.php?topic=3127974.0 ; https://www.asicminervalue.com/miners/baikal/bk-n ; https://cointelegraph.com/news/bitmain-announces-new-monero-mining-antminer-x3-cryptos-devs-say-will-not-work
- [S53] https://github.com/monero-project/monero/pull/3253 (v7); https://coinguides.org/monero-network-upgrade-v8-cnv2-beryllium-bullet/ ; https://github.com/SChernykh/CryptonightR (CN-R: chip latency up 2.5x)
- [S54] https://medium.com/@MoneroCrusher/analysis-more-than-85-of-the-current-monero-hashrate-is-asics-and-each-machine-is-doing-128-kh-s-f39e3dca7d78 ; https://beincrypto.com/hashrate-analysis-reveals-asics-account-for-85-of-monero-mining/
- [S55] https://github.com/tevador/randomx (30 Nov 2019)
- [S56] https://github.com/tevador/RandomX/blob/master/doc/design.md
- [S57] https://github.com/tevador/RandomX/blob/master/doc/specs.md
- [S58] https://xmrig.com/benchmark/5kFcJv (3950X); https://whattomine.com/coins/101-xmr-randomx/gpus (RTX 3090 at 2.0 kH/s, 290 W)
- [S59] https://bt-miners.com/products/bitmain-antminer-x5-monero-miner-212k-bt-miners/ ; https://bitmain.com.vc/news/bitmain-launches-antminer-x9 ; https://pineconeinibox.shop/product/pinecone-matches-inibox-r1x-xmr-edition/ ; https://github.com/xmrig/xmrig/blob/master/doc/ALGORITHMS.md ; https://rfc.tari.com/RFC-0131_Mining ; https://www.theblock.co/post/353240/tari-privacy-network-merged-monero-mining-launch-mainnet
- [S60] https://github.com/tevador/RandomX/blob/master/README.md (the four audits and their cost); https://blog.trailofbits.com/2019/07/02/state/
- [S61] https://github.com/mimblewimble/docs/blob/master/docs/about-grin/proof-of-work.md ; https://github.com/tromp/cuckoo/blob/master/README.md ; https://github.com/tromp/cuckoo/blob/master/doc/cuckoo.pdf
- [S62] https://forum.grin.mw/t/mid-july-pow-hardfork-cuckaroo29-cuckarood29/5082 ; https://www.cudominer.com/grin-network-update-hard-fork-16th-january-2020/ ; https://forum.grin.mw/t/grin-v5-0-0-network-upgrade-hard-fork-4-january-2021/7895
- [S63] https://docs.aeternity.com/aeternity-core-concepts/protocol/consensus-mechanisms/cuckoo-cycle-proof-of-work ; https://medium.com/cortexlabs/miners-can-now-test-mine-on-testnet-dolores-in-preparation-for-the-mainnet-launch-cf851d7b0146
- [S64] https://forum.grin.mw/t/introducing-the-grn1-a-cuckatoo31-asic-from-obelisk/2519 ; https://medium.com/obelisk-blog/grn1-cancellation-announcement-54782c6e3e83
- [S65] https://bitcointalk.org/index.php?topic=5219851.0 ; https://forum.grin.mw/t/innosilicons-grin-asics-canceled/6932 ; https://ipollo-miners.com/product/ipollo-g1/
- [S66] https://forum.grin.mw/t/another-cuckatoo-bounty-succesfully-claimed/11739 (Apr 2025)
- [S67] https://eips.ethereum.org/EIPS/eip-1057 ; https://github.com/ifdefelse/ProgPOW
- [S68] https://github.com/ethereum/pm/blob/master/AllCoreDevs-EL-Meetings/Meeting%2052.md ; https://www.coindesk.com/markets/2019/01/04/ethereum-developers-give-tentative-greenlight-to-asic-blocking-code ; https://souptacular.github.io/2020-03-02-progpow-the-ethereum-community-speaks/ ; https://www.coindesk.com/tech/2020/03/06/ethereums-progpow-call-features-frustration-but-little-progress
- [S69] https://leastauthority.com/static/publications/LeastAuthority-ProgPow-Algorithm-Final-Audit-Report.pdf (9 Sep 2019)
- [S70] https://github.com/ethcatherders/progpow-audit ("Bob Rao - ProgPOW Hardware Audit Report Final.pdf", Sep 2019)
- [S71] https://github.com/kik/progpow-exploit (4 Mar 2020); https://github.com/Souptacular/linzhi (Linzhi's 3x to 8x claim)
- [S72] https://cryptoage.com/en/1238-bitcoin-interest-bci-and-new-mining-algorithm-progpow.html ; https://en.wikipedia.org/wiki/Zano_(blockchain_platform) ; https://github.com/sero-cash/serominer ; https://x.com/QuaiNetwork/status/1880037240149057759 ; https://veil-project.com/blog/2020-OhGodAGirl/
- [S73] https://docs.ergoplatform.com/mining/autolykos/ ; https://ergoplatform.org/en/blog/2019_07_09_after_launch/ ; https://bytwork.com/en/news/khardfork-ergo-07
- [S74] https://www.hashrate.no/gpus/3090/ERG
- [S75] https://mining.confluxnetwork.org/ ; https://github.com/Conflux-Chain/conflux-rust
- [S76] https://github.com/Conflux-Chain/CIPs/blob/master/CIPs/cip-102.md
- [S77] https://www.kaspafaq.com/sp_accordion_faqs/what-is-kheavyhash/ ; https://github.com/Dagmbisrat/Kaspa-FPGA-Miner
- [S78] https://whattomine.com/coins/352-kas-kheavyhash/gpus/49-nvidia-geforce-rtx-3090
- [S79] https://www.cryptominerbros.com/product/iceriver-ks0-100gh-s-kas-miner/ ; https://www.asicminervalue.com/miners/iceriver/ks1 ; https://apextomining.com/product/new-bitmain-antminer-ks3-8-3t-3188w-kas-miner-asic-mining-machine-profitable-comining-soon/ ; https://www.asicminervalue.com/miners/bitmain/antminer-ks5-pro-21th
- [S80] https://hashdag.medium.com/kaspa-where-to-part-iv-last-c68717a8d309 (May 2023); https://miningreturns.com/news/kaspa-asic-mining-era-what-you-need-to-know
- [S81] https://x.com/karlsennetwork/status/1829148683104870534 ; https://www.hashrate.no/c/Algorithm_change_for_Karlsen_and_Pyrin ; https://github.com/spectre-project/rusty-spectre ; https://cryptix-network.org/whitepaper ; https://network.hoosat.fi/public/htn-whitepaper-2.pdf ; https://bugna.org/
- [S82] https://spec.nexa.org/mining/NexaPOW/
- [S83] https://www.cryptominerbros.com/product/dragonball-miner-a21-nexa-miner/ ; https://whattomine.com/coins/357-nexa-nexapow
- [S84] https://docs.alephium.org/frequently-asked-questions/ ; https://medium.com/@alephium/one-year-of-mainnet-b7ed5d3024ee
- [S85] https://hashrate.no/gpus/3090/ALPH
- [S86] https://www.asicminervalue.com/miners/goldshell/al-box ; https://mineshop.eu/bitmain-antminer-al1 ; https://www.zeusbtc.com/Asic-Miner/Asic-Miner-Details.asp?ID=3719
- [S87] https://fips.ironfish.network/fips/fip-3-memory-hard-mining-algorithm ; https://fips.ironfish.network/fips/fip-10-hardfork-1 ; https://github.com/iron-fish/fish-hash ; https://github.com/karlsen-network/fish-hash-plus
- [S88] https://medium.com/nervosnetwork/a-decentralized-mainnet-launch-for-nervos-ckb-9cb119d15540
- [S89] https://www.asicminervalue.com/miners/bitmain/antminer-k5-1130gh ; https://www.asicminervalue.com/miners/goldshell/ck5 ; https://2miners.com/blog/nervos-ckb-network-hashrate-increased-asics-are-the-cause/
- [S90] https://www.coindesk.com/markets/2020/02/04/handshakes-uncensorable-web-domains-go-live-on-mainnet
- [S91] https://www.goldshell.com/news/goldshell-announces-best-handshakehns-miner-hs1-coming-soon/ ; https://www.asicminervalue.com/miners/goldshell/hs3
- [S92] https://radiantblockchain.org/ ; https://d-central.tech/miners/rxd-rx0/
- [S93] https://cryptoage.com/en/2929-video-card-hashrate-based-on-the-sha512-256d-algorithm-cryptocurrency-mining-radiant-rxd.html
- [S94] https://cryptomining-blog.com/11865-mining-zano-using-the-progpowz-proof-of-work-algorithm/
- [S95] https://github.com/dynexcoin/DynexSolve ; https://minerstat.com/coin/DNX/faq
- [S96] https://docs.warthog.network/janushash/ ; https://github.com/CoinFuMasterShifu/Janushash
- [S97] https://docs.xelis.io/network-upgrades
- [S98] https://docs.verus.io/overview/verus-proof-of-power.html
Papers:
- [P1] https://www.microsoft.com/en-us/research/publication/moderately-hard-memory-bound-functions/
- [P2] https://www.wisdom.weizmann.ac.il/~naor/PAPERS/mem.pdf
- [P3] https://www.tarsnap.com/scrypt/scrypt.pdf
- [P4] https://eprint.iacr.org/2014/238
- [P5] https://eprint.iacr.org/2016/115
- [P6] https://eprint.iacr.org/2016/759
- [P7] https://eprint.iacr.org/2016/989
- [P8] https://www.cryptolux.org/images/d/d0/Argon2ESP.pdf
- [P9] https://eprint.iacr.org/2016/027
- [P10] https://eprint.iacr.org/2015/227
- [P11] https://eprint.iacr.org/2013/525
- [P12] https://eprint.iacr.org/2015/136
- [P13] https://eprint.iacr.org/2017/225
- [P14] https://eprint.iacr.org/2018/221
- [P15] https://eprint.iacr.org/2017/443
- [P16] https://eprint.iacr.org/2015/946
- [P17] https://arxiv.org/abs/1606.03588
- [P18] https://eprint.iacr.org/2017/497
- [P19] https://eprint.iacr.org/2014/059
- [P20] https://da-data.blogspot.com/2014/03/a-public-review-of-cuckoo-cycle.html ; http://www.cs.cmu.edu/~dga/crypto/cuckoo/analysis.pdf
- [P21] https://arxiv.org/abs/1902.00112
- [P22] https://ieeexplore.ieee.org/document/8516911/
- [P23] http://www.vldb.org/pvldb/vol13/p898-feng.pdf
- [P24] https://arxiv.org/abs/2002.11064
Economics and silicon:
- [E1] https://europractice-ic.com/schedules-prices-2025/
- [E2] https://newsletter.semianalysis.com/p/the-dark-side-of-the-semiconductor (24 Jul 2022)
- [E3] https://semiengineering.com/big-trouble-at-3nm/ (21 Jun 2018); https://semiengineering.com/the-increasingly-uneven-race-to-3nm-2nm/ (24 May 2021); https://semiengineering.com/what-will-that-chip-cost/ (30 Oct 2023)
- [E4] https://siliconanalysts.com/analysis/fabless-startup-tapeout-cost-guide (1 Mar 2026, secondary)
- [E5] https://patentpc.com/blog/chip-manufacturing-costs-in-2025-2030-how-much-does-it-cost-to-make-a-3nm-chip (secondary, approximate)
- [E6] https://techcrunch.com/2018/09/26/bitmain-hong-kong-ipo/ ; https://bitcoinmagazine.com/markets/bitmain-ipo-prospectus-reveals-offering-may-be-gamble-investors ; https://www.chaincatcher.com/en/article/2057998
- [E7] https://www.sec.gov/Archives/edgar/data/1780652/000119312519297270/d773846d424b4.htm
- [E8] https://michaeltaylor.org/papers/bitcoin_taylor_cases_2013.pdf ; https://michaeltaylor.org/papers/Taylor_Bitcoin_IEEE_Computer_2017.pdf
- [E9] https://www.tomshardware.com/news/amd-unveils-more-ryzen-3d-packaging-and-v-cache-details-at-hot-chips (Aug 2021); https://www.graphcore.ai/posts/introducing-second-generation-ipu-systems-for-ai-at-scale ; https://www.theregister.com/software/2020/09/29/groq-is-hard-to-grok-but-reckons-its-ai-chips-roq-ex-googlers-unorthodox-design-now-shipping-to-customers/1170931
- [E10] https://newsletter.semianalysis.com/p/tsmcs-3nm-conundrum-does-it-even (21 Dec 2022); https://fuse.wikichip.org/news/7343/iedm-2022-did-we-just-witness-the-death-of-sram/ ; https://www.tomshardware.com/news/no-sram-scaling-implies-on-more-expensive-cpus-and-gpus ; https://images.nvidia.com/aem-dam/Solutions/geforce/blackwell/nvidia-rtx-blackwell-gpu-architecture.pdf (GB202: 128 MB L2 on the full die, 96 MB on the RTX 5090, 750 mm^2)
- [E11] CoinMarketCap historical snapshots, https://coinmarketcap.com/historical/YYYYMMDD/ for the dates in the section 2.5 table (pulled 5 Oct 2026); HBM pricing https://www.trendforce.com/presscenter/news/20240506-12125.html and https://www.nextplatform.com/2024/02/27/he-who-can-pay-top-dollar-for-hbm-memory-controls-ai-training/ ; the E3's DDR3 and the A10's presumed GDDR6 from the ProgPoW FAQ https://medium.com/@ifdefelse/progpow-faq-6d2dce8b5c8b (Jan 2019, read through secondary coverage) and https://coingeek.com/memory-limitations-prompt-bitmain-antminer-e3-to-halt-etc-support/
Latency:
- [L1] https://terpconnect.umd.edu/~blj/papers/memsys2018-dramsim.pdf
- [L2] https://arxiv.org/abs/1712.08304
- [L3] https://docs.nvidia.com/nsight-compute/ProfilingGuide/index.html ; https://developer.download.nvidia.com/video/gputechconf/gtc/2020/presentations/s21819-optimizing-applications-for-nvidia-ampere-gpu-architecture.pdf
- [L4] https://arxiv.org/abs/2604.15751

View file

@ -0,0 +1,85 @@
# Card lifetime per tier: how many years a card keeps mining
5 October 2026. Consequences review, sub-agent of the consequences reviewer. Desk arithmetic only; nothing was run.
## 1. Inputs
| Input | Source | Value used |
|---|---|---|
| Dataset schedule | `docs/spec/01-lottery-hash.md` 432 to 437 | 2 GiB at genesis plus 0.5 GiB a year (2,048 + 512 x years MiB) |
| Index mapping (a) | same file, line 442 | multiply-shift: the dataset grows every day, continuous |
| Index mapping (b) | same file, line 442 | power-of-two steps 2, 4, 8 GiB on the schedule's average: 4 GiB at year 4, 8 GiB at year 12; my extrapolation: 16 GiB at year 28, 32 GiB at year 60 |
| Scratch per resident warp | `igneum-wt-ca2-cache/docs/plans/hot-table.md` 66 to 73 | 32 or 128 KiB per warp; 5090 = 170 SMs x 48 warps = 8,160 (approximate, from memory) |
| Hot table, buffers | same file, 70 | hot table 32, 64 or 96 MiB (96 used here); buffers 128 MiB |
| Cache | hot-table.md 70 (resident, 256 MiB in every total) against `igneum-wt-ca2-era/docs/plans/era-layout.md` 93 ("resident only while the day's dataset is built, then free") | both readings carried: resident = worst case, freed = best case. The two plans disagree and gate 1 should say which |
| Cache growth | `igneum-wt-ca2-coord/docs/plans/counter-asic-2-status.md` 79 (layer 6 option C) | 256 MiB at genesis, 512 MiB at year 4, 1 GiB at year 12; by the same rule 2 GiB at year 28, 4 GiB at year 60 |
| Budget rule | same file, 17: the whole working set stays under 6 GB on an 8 GB card | my reading: 75% of card memory at every tier. Apple: 50% of unified memory, because macOS, the display and the node share it; that share is my assumption |
| Public claims | `site/index.html` 443, 461; `site/litepaper.html` 560; `docs/evidence.md` | quoted in Table 3. evidence.md has no row on card lifetime |
Card memory is binary (8 GB = 8,192 MiB). The hot-table row "An 8 GB card at 5090 occupancy" (line 73) counts 8,160 warps; a real 8 GB card has 20 to 24 SMs, so its scratch is about a tenth of that row. Resident warps below are SMs x 48 (NVIDIA Ampere and later), SM counts from memory, approximate; Apple uses the 2,048 warps the Metal harness launches (hot-table.md 66).
## 2. Table 1: non-dataset working set per tier (MiB)
Worst = scratch 128 KiB, cache resident. Columns g / y4 / y12 = genesis, year 4, year 12 (the cache doublings). Freed = era-layout's reading, constant over the years.
| Tier | Card assumed (SMs, approximate) | Warps | Scratch 128 KiB | Scratch 32 KiB | Cache resident, 128 KiB: g / y4 / y12 | Cache resident, 32 KiB: g / y4 / y12 | Cache freed: 128 / 32 KiB |
|---|---|---|---|---|---|---|---|
| 4 GB | GTX 1650 (14 SMs x 32 warps, Turing) | 448 | 56 | 14 | 536 / 792 / 1,304 | 494 / 750 / 1,262 | 280 / 238 |
| 8 GB | RTX 3050 (20) | 960 | 120 | 30 | 600 / 856 / 1,368 | 510 / 766 / 1,278 | 344 / 254 |
| 12 GB | RTX 3060 (28) | 1,344 | 168 | 42 | 648 / 904 / 1,416 | 522 / 778 / 1,290 | 392 / 266 |
| 16 GB | RTX 5060 Ti (36) | 1,728 | 216 | 54 | 696 / 952 / 1,464 | 534 / 790 / 1,302 | 440 / 278 |
| 24 GB | RTX 4090 (128) | 6,144 | 768 | 192 | 1,248 / 1,504 / 2,016 | 672 / 928 / 1,440 | 992 / 416 |
| 32 GB | RTX 5090 (170) | 8,160 | 1,020 | 255 | 1,500 / 1,756 / 2,268 | 735 / 991 / 1,503 | 1,244 / 479 |
| Apple 8 to 64 GB | M-series, harness launch count | 2,048 | 256 | 64 | 736 / 992 / 1,504 | 544 / 800 / 1,312 | 480 / 288 |
Every row = scratch + 96 (hot table) + 128 (buffers) + cache (256 / 512 / 1,024 when resident). The freed reading still peaks at dataset + cache during the daily build, but that peak is smaller than the resident total whenever hashing pauses for the build, so the freed column is the steady-state set.
## 3. Table 2: dataset room and the year the dataset outgrows it
Room = usable memory (75%, Apple 50%) minus Table 1. Worst = 128 KiB scratch, cache resident (room shrinks at years 4, 12, 28, 60). Best = 32 KiB scratch, cache freed. Option (a): the year 2,048 + 512 x y exceeds the room. Option (b): the first step the room cannot hold; the card mines up to that day.
| Tier | Usable MiB (share) | Room at genesis, worst / best | (a) ends, years, worst / best | (b) ends, year, worst / best |
|---|---|---|---|---|
| 4 GB | 3,072 (75%) | 2,536 / 2,834 | 1.0 / 1.5 | 4 / 4 |
| 8 GB | 6,144 (75%) | 5,544 / 5,890 | 6.3 / 7.5 | 12 / 12 |
| 12 GB | 9,216 (75%) | 8,568 / 8,950 | 12.0 / 13.5 | 12 / 28 |
| 16 GB | 12,288 (75%) | 11,592 / 12,010 | 17.1 / 19.5 | 28 / 28 |
| 24 GB | 18,432 (75%) | 17,184 / 18,016 | 28.0 / 31.2 | 28 / 60 |
| 32 GB | 24,576 (75%) | 23,076 / 24,097 | 37.6 / 43.1 | 60 / 60 |
| Apple 8 GB | 4,096 (50%) | 3,360 / 3,808 | 2.6 / 3.4 | 4 / 4 |
| Apple 16 GB | 8,192 (50%) | 7,456 / 7,904 | 10.1 / 11.4 | 12 / 12 |
| Apple 32 GB | 16,384 (50%) | 15,648 / 16,096 | 25.1 / 27.4 | 28 / 28 |
| Apple 64 GB | 32,768 (50%) | 32,032 / 32,480 | 55.1 / 59.4 | 60 / 60 |
What the table says per tier:
| Tier | Reading |
|---|---|
| 4 GB | Mines at genesis with 488 to 786 MiB spare. Under (a) it is out within 1 to 1.5 years. Under (b) it lasts to the year-4 step, as the spec's own remark says (line 442) |
| 8 GB | 6 to 7.5 years under (a). 12 years under (b): "more than a decade" is true only under (b), and only just |
| 12 GB | The year-12 cache doubling (1 GiB resident) is what ends it, under both options, if the cache stays resident. With the cache freed it reaches year 28 under (b). This tier's lifetime is decided by the cache residency question, not by the dataset |
| 16 GB | 17 to 19.5 years under (a), year 28 under (b) |
| 24 GB | Under the resident reading the year-28 cache doubling (2 GiB) ends it the same day under both options. Freed: 31 years or year 60 |
| 32 GB | 38 to 43 years under (a), year 60 under (b). Not a constraint for any plan |
| Apple 8 GB | 2.6 to 3.4 years under (a), year 4 under (b). The base 8 GB Apple laptop is a short-lived miner |
| Apple 16 GB | 10 to 11.4 years under (a), year 12 under (b): the same shape as an 8 GB card |
| Apple 32 / 64 GB | 25 years and 55 years or more. No constraint |
Proving is a separate budget (the 15.6 GB peak the 12 GB mine-and-prove question came from); this file covers mining only.
## 4. Table 3: the public sentences against the numbers
| Where | Sentence now | What the tables give | Proposed sentence (the project lead decides the wording) |
|---|---|---|---|
| `site/index.html` 443 | Memory: "2 GB, fixed" (RandomX) / "2 GB, growing" (Igneum) | 2 GiB at genesis, plus 0.5 GiB a year on average under either option | "2 GB, growing 0.5 GB a year". The row is right; the rate is the useful addition |
| `site/index.html` 461 | "Any 4 GB card, approximate." | True at genesis (2,584 to 2,834 MiB of a 3,072 MiB budget). Ends at 1 to 1.5 years under (a), year 4 under (b) | "Any 4 GB card at launch, 8 GB for the long run, approximate." |
| `site/litepaper.html` 560 | "a 4 GB card mines for about four years and an 8 GB card for more than a decade, approximate." | 4 GB: 1 to 1.5 years (a) or 4 years (b). 8 GB: 6.3 to 7.5 years (a) or 12 years (b). Both numbers hold only under option (b) | If gate 1 picks (b): "a 4 GB card mines until the first dataset step at year 4, an 8 GB card until the second at year 12 and a 16 GB card until year 28, approximate." If (a): "a 4 GB card mines for about a year, an 8 GB card for about seven and a 16 GB card for about seventeen, approximate." |
| `site/litepaper.html` 560 | "12 GB or more proves full shards." | Not a lifetime claim; left as is. For mining, 12 GB lasts 12 years with the cache resident, year 28 with it freed under (b) | No change from this file |
| `docs/evidence.md` | No row on card lifetime | The litepaper sentence is a public claim with no row | Add a row, label "designed", sources: spec 1.13.3 and this file; status moves to "tested" once a 4 GB and an 8 GB card run the genesis working set under the cap |
## 5. Reading
- The two index-mapping options end on the same day where a cache doubling takes the last of the room. With the cache resident that is the 12 GB tier at year 12 and the 24 GB tier at year 28 (Table 2, worst column). Under option (b) every tier ends on a step day by construction, so a tier ends on the same day under both options exactly when option (a) also ends it on a doubling day.
- Everywhere else option (b) is kinder: 4 GB gains about 2.5 years, 8 GB about 5, 16 GB about 10. The site and litepaper numbers are option (b) numbers. If gate 1 picks (a), both public sentences are wrong today by 2.5 to 5 years.
- The cache residency disagreement (hot-table.md 70 against era-layout.md 93) decides the 12 GB tier's lifetime (12 against 28 years) and nothing else. It should be settled at gate 1 beside the mapping choice.
- The 75% rule is my generalisation of "under 6 GB on an 8 GB card"; at 4 GB it leaves 1 GB for the driver and the display, which a headless rig would not need. A 4 GB card on a bare Linux rig might hold out to year 2 under (a). Not measured.

View file

@ -0,0 +1,100 @@
# The on-die-cache recompute chip against the RTX 5090, class v2 and class v3, everything combined
5 October 2026 (night), Counter ASIC 2.0, worker ca2-mixer. The model is M16's
(`docs/analysis/m16-recompute-attacker-2026-10-05.md`): the strongest chip the plan has priced holds the whole
cache in SRAM and derives every dataset item instead of reading it, so its cost per hash is item derivations,
and its rate at a 50 T op/s integer budget (an RTX 5090's, approximate) is `50 T / (ops per hash)`. Nothing here
is a measurement of a chip; every GPU figure says where it was measured. "Approximate" marks a figure from memory.
## 1. Inputs
| Input | Value | Source |
|---|---|---|
| Items per hash | 128 (one item per load, 128 loads per hash, median 128.00 distinct) | spec 01 sections 1.4.2 and 1.8.5; the 20,000-program census |
| Integer operations per mixer application | about 130 | spec 01 section 1.8.4 |
| Mixer applications per item | 9 under v2; 36 under v3 (`m = 4`, `docs/plans/mixer-x4.md`) | `memhard::Shape::mixers_per_item` |
| Integer operations per item | 1,170 (v2); 4,680 (v3) | 9 x 130; 36 x 130 |
| Integer operations per hash | 149,760 (v2, "150,000"); 599,040 (v3, "600,000") | 128 x the above |
| Chip integer budget | 50 T op/s (approximate: 21,760 ALUs at about 2.4 GHz, one 32-bit operation each per clock) | M16 section 3 |
| Fixed-function factor | 3x (approximate, from memory: 2x to 5x is the usual credit for a pipeline with no scheduling or divergence) | M16 section 3 |
| RTX 5090, version 2 programs, measured | 136.1 MH/s (readwidth, tonight, `docs/plans/read-width.md`, pack w4 on PC 2); 139.7 MH/s (M11, 4 October, `docs/bench-log.md`) | this analysis uses tonight's 136.1 as the denominator and quotes both |
| RTX 5090 at w16 (16-byte loads), measured | 139.8 MH/s | readwidth table, tonight (the width stays 4 B: w16 closes nothing) |
| Cache mirror, 256 MiB, N5 headline density | 128 mm^2, $46 per good die (64 mm^2, $21 at the bit-cell lower bound) | `docs/analysis/sram-mirror.md` revision 2, sections 4 and 5 (`ca2-analysis` e6085c6) |
| Cache mirror plus a 96 MB hot table, N5 headline | 175 mm^2, $68 | same, so a hot table costs 0.49 mm^2 and $0.23 per MB (linear, approximate) |
| 512 MiB and 1 GiB mirrors, N5 headline | 255 mm^2 and 510 mm^2; $111 to $306 | same, section 4 (the growth rule's cache at years 4 and 12, priced at today's node) |
| GPU-class die | 750 mm^2 (the equal-silicon comparison) | M16 section 3 |
| CPU verifier, one M5 Max core (loaded, load average 5.6; ratios are the measurement) | v2 1.31 to 1.36 ms per unit, x4 1.92 to 1.96 (1.45x), x8 2.79 (2.1x); worst cold 1.58 / 2.04 / 2.94 ms | `docs/plans/mixer-x4.md` section 6.4, 5 October 2026 21:40 UTC |
## 2. The rows
Chip rate = 50 T op/s / ops per hash. "Bare" = chip rate / 136.1 MH/s. "With the factor" = bare x 3. "Equal
silicon" = bare x (750 - SRAM) / 750 x 3: the SRAM takes die area the logic does not get, the M16 convention
("minus the area the SRAM takes"). SRAM in mm^2 and dollars at the N5 headline density.
| Row | Mixer | Ops per hash | Chip rate at 50 T op/s | SRAM the chip holds | mm^2 / $ (N5 headline) | Bare gain against 136.1 MH/s | With the 3x factor | Equal silicon, SRAM deducted, with the factor |
|---|---|---|---|---|---|---|---|---|
| v2 as shipped (the M16 and scratch-soundness row) | x1 | 149,760 | 334 MH/s | 256 MiB | 128 / $46 | 2.45x (2.39x against 139.7) | 7.4x | 6.1x |
| v2 at w16 (not adopted; the chip's cost is items, not bytes: unchanged) | x1 | 149,760 | 334 | 256 MiB | 128 / $46 | 2.39x against 139.8 | 7.2x | 5.9x |
| x4 (the candidate measured beside v3; not v3) | x4 | 599,040 | 83.5 MH/s | 256 MiB | 128 / $46 | 0.61x | 1.84x | 1.53x |
| MEASURED, NOT ADOPTED (layer 5 decided out of v3 on the PC rows, coordinator 21:40 UTC): v3 plus a 32 MiB hot table, added form (16 dataset loads and k hot loads): the honest card pays the hot loads, this chip pays SRAM only | x4 | 599,040 (a hot load is one SRAM read, no item) | 83.5 | 288 MiB | 144 / $53 | 0.66x at the Mac's g = 0.93 (126.6 MH/s); 0.70x at the 5090's g = 0.87 (118.4); the 9070 XT's g 0.84 | 1.98x (Mac g), 2.11x (5090 g) | 1.60x, 1.71x |
| MEASURED, NOT ADOPTED: v3 plus a 64 MiB hot table, added form | x4 | 599,040 | 83.5 | 320 MiB | 160 / $61 | 0.71x at the Mac's g = 0.87 (118.4 MH/s); 0.73x at the 5090's g = 0.84 (114.3); the 9070 XT's g 0.80 | 2.12x (Mac g), 2.19x (5090 g) | 1.67x, 1.73x |
| v3 at year 4 (cache 512 MiB under option C, dataset 4 GiB), no hot table | x4 | 599,040 | 83.5 | 512 MiB | 255 / $111 | 0.61x | 1.84x | 1.21x |
| v3 at year 12 (cache 1 GiB, dataset 8 GiB) | x4 | 599,040 | 83.5 | 1 GiB | 510 / $306 | 0.61x | 1.84x | 0.59x |
| **v3: mixer x8** (decided 22:05 UTC under the delegated rule: verify 2.1 ms per unit on one Mac core against the 10 ms gate, the daily 1 GiB build 23 to 77 ms on the 5090 and the 9070 XT) | x8 | 1,198,080 | 41.7 | 256 MiB | 128 / $46 | 0.31x | 0.92x | 0.76x |
| x8 at year 4 | x8 | 1,198,080 | 41.7 | 512 MiB | 255 / $111 | 0.31x | 0.92x | 0.61x |
The era draws of spec 1.13.1 cost the chip nothing in this model: the mixer round count is not drawn, the op
weights and fold rotations change the program, not the item derivation, so the chip's ops per hash stand. The
width rule (4-byte loads kept) changes nothing either: w16 would have moved the honest denominator by 2.7% and the
chip's cost not at all.
Arithmetic, row v3: 36 x 130 = 4,680 ops per item; x 128 = 599,040 per hash; 50 x 10^12 / 599,040 = 83.5 x 10^6
hashes per second; 83.5 / 136.1 = 0.613; x 3 = 1.84; equal silicon (750 - 128) / 750 = 0.829, x 1.84 = 1.53.
Hot table rows: 32 MiB x 0.49 mm^2 per MB = 16 mm^2, 64 MiB = 32 mm^2 (the 96 MB column of `sram-mirror.md`
scaled linearly); (750 - 144) / 750 = 0.808 and (750 - 160) / 750 = 0.787. The honest denominator in the added
form is the v2 rate times `g`, the card's measured ratio with the hot loads added: on the M5 Max tonight
`g = 0.93 / 0.87 / 0.83` at 32 / 64 / 96 MiB (the cache agent, relayed by the coordinator at 21:23 UTC;
`docs/plans/hot-table.md` carries the runs); the 5090's and the 9070 XT's `g` are the PC rows, owed, and until they
land the row carries the Mac's `g` against the 5090's rate, which is a mixed figure and is marked so. Year 4 and 12 rows: the mirror of
`sram-mirror.md` section 4 at N5 for 512 MiB and 1 GiB plus the 64 MiB table, at today's density (the node of
those years is denser by about 1.8x at year 10 on the trend the same file cites; the row is a floor on the area,
not a forecast).
## 3. The margin, plainly
The combined headline row is the mixer row alone (layer 5 is out: the added form costs the 5090 13 to 16 percent
and the 9070 XT 16 to 20 percent against the 0.97 bar, coordinator 21:40 UTC; the width stays 4 bytes; the era
draws and the cache growth cost this chip nothing at year 0), and class v3 is x8 (decided 22:05 UTC). The headline:
**the on-die-cache recompute chip at 50 T op/s reaches 41.7 MH/s against the 5090's 136.1, 0.31x bare, 0.92x with
the 3x fixed-function factor, 0.76x with the mirror's area deducted: under 1x with the factor, 0.92x, a margin of 8
percent on the factor (a 3.3x factor reads 1.0x) and of 9 percent on the budget (55 T op/s reads 1.0x).** The x4
candidate, measured beside it, read 1.84x and 1.53x. The hot-table rows above are kept as measured, not adopted:
against THIS chip an added hot table is a cost to the honest card and none to the chip, so it would have moved the
row the wrong way by the card's own `g`. The margin, plainly:
- the 3x fixed-function factor is approximate and from memory; at 3.3x the equal-budget row reads 2.0x;
- the denominator is one card's measured rate on one night (136.1 against 139.7 the night before: 2.6% apart);
- the 50 T op/s budget is approximate; a chip at 55 T op/s reads 2.0x;
- the hot table in the added form lowers the honest denominator by whatever the hot loads cost the GPU (owed from
the PC rows), which raises the chip's gain by the same share, 1.84x or more if the hot loads are free, higher if
not; the hot table's only cost to this chip is 16 to 32 mm^2 of die.
What keeps it under 1x is the mixer, and nothing else in Counter ASIC 2.0 moves this chip (the scratch at any share
gave 2.4x, `docs/analysis/scratch-soundness.md` section 3.4; the hot table taxes the DRAM-only chip, not this one;
the cache growth taxes it only in die area, which is cheap at year 0 and real at year 12). The next levers, in
order:
1. Mixer x16 (the next step of the same lever): 0.16x bare and 0.46x with the factor against 136.1; the verifier
by the measured increments (+0.63 ms at x4, +1.46 at x8 on the M5 Max core: about +3.1 ms at x16, 3.7 ms per
unit, 9 ms on a 2.5x slower laptop core, approximate) is at the edge of the 10 ms gate, so a 2019-class laptop
core measurement (O-1.14) decides it, not this model.
2. The hot table: adopted or not on the PC rows (`docs/plans/hot-table.md`); in the added form it costs the GPU
7 to 17 percent on the Mac and the chip die area only, so against this chip it is a lever in the wrong
direction and against a DRAM-only chip the first lever; if it is adopted, the mixer must carry the extra `1/g`
(x8 at g = 0.87 reads 1.06x at the equal budget, 0.84x with the SRAM deducted).
## 4. What this does not settle
The items of M16 section 5 stand: the inline kernel on NVIDIA with a 64 MiB cache inside L2 (a measured point
under the "50 T op/s" row) is a PC job not yet run; the time-memory curve (O-1.6) is not drawn; the mixer has had
no cryptanalysis, and a shortcut inside it cuts the 4,680 directly; no chip has been priced beyond its SRAM.

View file

@ -0,0 +1,175 @@
# Layer 7: the integer matrix family (INT8 x INT8 into INT32) as a reserved instruction family, design
5 October 2026 (night), Counter ASIC 2.0 (`docs/plans/counter-asic-2.md`, layer 7), branch `ca2-analysis`. Design
only: nothing here touches the generator, a vector or a node. Every figure is cited (vendor document, URL, section) or
measured (machine, date, command) or labelled approximate.
## 1. The primitive per vendor, from the vendor documents
| Vendor, hardware | Per-lane dot4 (4 bytes x 4 bytes into a 32-bit integer) | Warp or wave matrix (int8 tiles, int32 accumulate) | Source |
|---|---|---|---|
| NVIDIA, sm_61 and later (Pascal on) | PTX `dp4a.atype.btype d, a, b, c` with `.atype = .btype = {.u32, .s32}`: "Four-way byte dot product which is accumulated in 32-bit result"; semantics `d = c; for i in 0..3: d += Va[i] * Vb[i]` with the bytes sign- or zero-extended by type; introduced in PTX ISA 5.0, "Requires sm_61 or higher". CUDA: `__device__ int __dp4a(int srcA, int srcB, int c)` ("Four-way signed int8 dot product with int32 accumulate") and the unsigned form, plus `char4`/`uchar4` overloads | `mma.sync` with `.u8`/`.s8` A and B and `.s32` C and D: shape `.m8n8k16` "requires sm_75 or higher" (Turing on, PTX 6.5); shapes `.m16n8k16` and `.m16n8k32` require sm_80 (Ampere on, PTX 7.0); sparse `.m16n8k32` and `.m16n8k64` with `.u8`/`.s8` also exist | PTX ISA 9.4, section 9.7.1.24 (dp4a) and 9.7.16.5 (mma), https://docs.nvidia.com/cuda/parallel-thread-execution/index.html ; CUDA Math API, integer intrinsics, https://docs.nvidia.com/cuda/cuda-math-api/cuda_math_api/group__CUDA__MATH__INTRINSIC__INT.html ; read 5 October 2026 |
| AMD RDNA 3 (gfx11) | `v_dot4_i32_iu8` (VOP3P; each operand signed or unsigned by a per-operand bit, optional clamp) reached from clang/HIP/OpenCL C as `__builtin_amdgcn_sudot4(bool a_signed, int a, bool b_signed, int b, int acc, bool clamp)` (LLVM feature `dot8-insts`: "Has v_dot4_i32_iu8, v_dot8_i32_iu4 instructions"); `v_dot4_u32_u8` as `__builtin_amdgcn_udot4` (`dot7-insts`: "Has v_dot4_u32_u8, v_dot8_u32_u4"); `v_dot4_i32_i8` as `__builtin_amdgcn_sdot4` (`dot1-insts`: "Has v_dot4_i32_i8 and v_dot8_i32_i4"). gfx11's common feature set carries dot7, dot8, dot9, dot10 and dot12 | `V_WMMA_I32_16X16X16_IU8`: `__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32` and `_w64` (feature `wmma-256b-insts`), a 16x16x16 tile per wave | LLVM `clang/include/clang/Basic/BuiltinsAMDGPU.td` (main, read 5 October 2026), lines defining `__builtin_amdgcn_sdot4`, `udot4`, `sudot4`, `wmma_i32_16x16x16_iu8_w32`; AMD GPUOpen, "How to accelerate AI applications on RDNA 3 using WMMA", https://gpuopen.com/learn/wmma_on_rdna3/ ; the RDNA 3 ISA PDF itself did not download tonight (AMD's CDN refused curl and the fetcher timed out), so the instruction names are from the compiler and GPUOpen, not quoted from the ISA guide |
| AMD RDNA 4 (gfx12, the 9070 XT) | the same `sudot4` and `udot4` builtins: LLVM's `FeatureISAVersion12_Generic` carries `FeatureDot7Insts` and `FeatureDot8Insts` and not `FeatureDot1Insts`, so `__builtin_amdgcn_sdot4` is NOT exposed on gfx12 and `sudot4` with both operands signed is the signed form to use | `__builtin_amdgcn_wmma_i32_16x16x16_iu8_w32_gfx12` and `_w64_gfx12` (feature `wmma-128b-insts`): the int8 WMMA exists on RDNA 4 with a narrower per-lane operand (2 ints per lane for A and B against 4 on RDNA 3); AMD's RDNA 4 WMMA guide names the same builtin | LLVM `llvm/lib/Target/AMDGPU/AMDGPU.td` (`FeatureISAVersion12_Generic`) and `BuiltinsAMDGPU.td` (main, 5 October 2026); AMD GPUOpen, "WMMA guide for AMD RDNA 4 architecture GPUs, part 2", https://gpuopen.com/learn/wmma-guide-amd-rdna-4-gpus-part-2/ ; the RDNA 4 ISA guide (AMD document 70651, April 2025) was not readable tonight (docs.amd.com returned 401 to a direct fetch) |
| AMD CDNA 3 (MI300) | the same VOP3P dot instructions (approximate: not checked in the CDNA 3 guide tonight) | `V_MFMA_I32_16X16X32_I8` and `V_MFMA_I32_32X32X16_I8` (opcodes 87 and 86 in the VOP3P-MFMA table) | AMD Instinct MI300 CDNA 3 ISA Reference Guide, 5 August 2025, https://www.amd.com/content/dam/amd/en/documents/instinct-tech-docs/instruction-set-architectures/amd-instinct-mi300-cdna3-instruction-set-architecture.pdf (downloaded and grepped 5 October 2026) |
| AMD, OpenCL on Adrenalin (Windows) | `cl_khr_integer_dot_product` is NOT in the 24 extensions Adrenalin lists for gfx1201 (the list: fp64, the int32 and int64 atomics, 3d image writes, byte addressable store, fp16, gl sharing, amd device attribute query, amd media ops and media ops2, d3d10, d3d11 and dx9 sharing, image2d from buffer, subgroups, gl event, depth images, mipmap image and writes, amd copy buffer p2p); the platform is OpenCL 2.1 so the OpenCL C 3.0 feature macro `__opencl_c_integer_dot_product_input_4x8bit` is not expected. What IS reachable: the Adrenalin OpenCL compiler is clang (driver string `PAL,LC`), and `__builtin_amdgcn_sudot4` from OpenCL C has been shown to emit `V_DOT4_I32_IU8` on a Radeon 780M (gfx1103, RDNA 3, driver 32.0.31041, Windows 11) at 2.7x the scalar fallback (1.27 to 3.42 TMAC/s) | not from OpenCL C | Adrenalin 26.9.2 extension list, https://geeks3d.com/20260904/amd-radeon-adrenalin-26-9-x-graphics-driver/ ; the OpenCL C route: https://github.com/1640675651/CPPminer/pull/1 (third party, one machine; the 9070 XT run of this document's probe is the check) ; `cl_khr_integer_dot_product` itself: OpenCL C 3.0 specification section 6.2.2.16, `int dot(char4, char4)` and `int dot_acc_sat(char4, char4, int)`, https://registry.khronos.org/OpenCL/specs/3.0-unified/html/OpenCL_Ext.html |
| Apple, Metal (MSL 4.1, 4 June 2026) | none. MSL has no dp4a or packed byte dot product: the built-in `dot(T x, T y)` is a geometric function on floating-point vectors (section 6.9); the integer functions of section 6.4 have no dot form. A per-lane dot4 is scalar emulation (section 4 below measures it) | `simdgroup_matrix<T, 8, 8>` exists for T = half, bfloat (Metal 3.1 and later) and float only (section 2.4: "T is half, bfloat ... or float"); no integer SIMD-group matrix. BUT Metal 4's tensor operation `mpp::tensor_ops::matmul2d` (section 7.2.1, table 7.3, "MatMul2D data type supported") lists A `char` x B `char` into C `int` (Metal 4) and `uchar` x `uchar` into `int` (Metal 4 and OS 26.4), plus `char` x `int4b_format` into `int`. So Apple has an exact int8 x int8 into int32 matrix path, on tensors (device or threadgroup memory, or a `cooperative_tensor` per SIMD-group or threadgroup), not on registers, and only through Metal 4's tensor API. Which GPU families run it in hardware (the M5's neural accelerators) against emulation is in the Metal Feature Set Tables, which the spec defers to and which were not read tonight | Metal Shading Language Specification version 4.1, https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf , sections 2.4, 6.9, 7.2.1 table 7.3 (PDF downloaded and text-extracted 5 October 2026) |
The Apple finding, stated plainly: the brief's expectation ("Apple has no int8 matrix or dot path") is half right. There
is no per-lane dot4 and no integer `simdgroup_matrix`. There is an exact `char x char -> int` matmul2d in Metal 4
(table 7.3). Two things about it are unverified tonight and matter for conformance: whether the int accumulate wraps or
saturates (the spec text I extracted says nothing either way; a vector at the int32 edge on the M5 settles it), and the
feature-set table (which Apple GPUs run it natively). What is settled: a generator op that is a per-lane dot4 has no
Apple intrinsic and costs scalar emulation; a generator op that is a whole-unit 8x8x16 or 16x16x16 int8 tile has a
native path on all three vendors (mma.sync on sm_75+, WMMA on RDNA 3 and 4, matmul2d on Metal 4), with Apple's path
living in a different API shape (tensors, not register fragments).
## 2. The family's semantics as a generator op (integer only, bit-exact)
Two forms are proposed; the reserve can hold both as separate families or one.
### 2.1 `dot4`: per-lane
```
dot4 dst = dst + dot4_u8(src, src2)
where dot4_u8(a, b) = sum over i in 0..3 of byte_i(a) * byte_i(b), bytes zero-extended, sum modulo 2^32
```
- Bytes are UNSIGNED. Reason, measured below: on Apple the unsigned emulation costs 1.6 ALU-chain steps per dot4
and the signed one 4.7 (section 4), while NVIDIA (`dp4a.u32.u32`) and AMD (`V_DOT4_U32_U8`, `udot4`, `dot7-insts`)
carry the unsigned form natively as they carry the signed one. Signed bytes buy nothing for the hash (the input is a
pseudo-random register) and cost the vendor without the intrinsic 3x more.
- Accumulation wraps modulo 2^32 like every other op in section 1 (spec 1.14 item 5). The maximum dot of four unsigned
bytes is 4 x 255 x 255 = 260,100, so no single dot4 overflows; the wrap is in the running sum, which is why the
AMD `clamp` bit and the OpenCL `dot_acc_sat` form are NOT the primitive (saturation would change results).
- Operands: `dst`, `src`, `src2` with `src != dst` as for `mad`; `src2` may equal either.
- Verifier: one closed-form integer expression per lane; the register-major interpreter of 1.11 adds four byte
multiplies and adds per lane. The CPU reference in the probes (`dot4_ref` in `proto-opencl/dot4-probe.c`) is this
expression.
### 2.2 `mm8`: the 32-lane unit as one int8 tile
The 32 lanes of a unit (spec 1.9) hold, in `src`, a 4-byte row fragment of an 8 x 16 int8 matrix A and, in `src2`,
a 4-byte column fragment of a 16 x 8 int8 matrix B, in exactly the layout of PTX `mma.m8n8k16` with `.u8` operands
(PTX ISA 9.4 section 9.7.16.5, "Matrix Fragments for mma.m8n8k16", the integer-type layout):
```
lane l (0..31): A[row = l >> 2][k = 4 * (l & 3) .. 4 * (l & 3) + 3] = the 4 bytes of src (byte 0 = lowest k)
B[k = 4 * (l & 3) .. +3][col = l >> 2] = the 4 bytes of src2
result C[r][c] = sum over k in 0..15 of A[r][k] * B[k][c] (uint8 x uint8, 16 products, exact, at most 1,040,400)
mm8 dst = dst + C[l >> 2][2 * (l & 3) + bit] bit = an immediate 0 or 1 drawn by the generator
```
Every lane receives one of the two C elements its lane position owns in the PTX fragment (`c0` for bit 0, `c1` for
bit 1), added into `dst` modulo 2^32. The whole op is a function of the unit's `src` and `src2` across all 32 lanes,
like `shfl`, so it needs the unit to be exactly 32 logical lanes (the wave64 rule of 1.9 applies: a wave64 device
holds two units and the local-memory path is used).
How each vendor runs it:
| Vendor | Native form | Cost per `mm8` (approximate until measured) |
|---|---|---|
| NVIDIA sm_75+ | one `mma.sync.aligned.m8n8k16.row.col.s32.u8.u8.s32` per warp, A and B fragments straight from `src` and `src2`, C = 0 in, `c0`/`c1` out, one add | one tensor instruction plus one add |
| AMD RDNA 3 and 4 | one `V_WMMA_I32_16X16X16_IU8` per wave32 with the 8x16 and 16x8 tiles zero-padded into 16x16 (the WMMA fragment layout differs from PTX's: a fixed permutation of bytes between lanes, which is a few `ds_bpermute` or `v_perm` operations, bit-exact) | one WMMA plus the permutation and the pad |
| AMD CDNA | `V_MFMA_I32_16X16X32_I8` with padding | as above |
| Apple, Metal 4 | `matmul2d<descriptor(8, 8, 16)>` on `uchar` A and B into an `int` cooperative tensor (table 7.3 row "uchar, uchar, int", OS 26.4), the fragments written from registers into a threadgroup tensor first (32 lanes x 8 bytes = 256 bytes), the C element read back per lane | one tensor op plus two threadgroup round trips; on Apple GPUs without the neural accelerators the runtime's emulation, unmeasured |
| Any vendor, fallback | 16 scalar byte products per lane after gathering the 16 bytes of B's column from the 4 lanes that hold them (4 shuffles or one 64-byte threadgroup exchange) | 4 shuffles plus 4 `dot4` emulations: on Apple about 4 x 1.6 = 6.4 ALU steps plus the shuffles (approximate, from the probe) |
Verifier: the unit evaluates C as 8 x 8 x 16 = 1,024 unsigned byte products once per `mm8` instruction and hands
each lane its element. That is 1,024 multiply-adds per instruction per unit, against 64 x 8 = 512 instructions per
hash: at W_new = 4 (section 3) a program carries about 2.6 `mm8` per iteration, 21 per hash, 21,500 multiply-adds per
unit per hash, under 10 microseconds on one core (approximate), far inside the 0.63 ms the verifier already spends per
unit (spec 1.11). The simulation stays exact because every product and sum is an integer with a defined wrap.
### 2.3 Which form to reserve
`mm8` is the one that takes matrix hardware at GPU scale from a chip (the plan's layer 7 row): a chip without tensor
units pays 1,024 products per unit per instruction where a GPU pays one tensor instruction. `dot4` is a per-lane ALU op
that a chip matches with four 8-bit multipliers, which is cheap silicon; it adds little chip resistance and costs Apple
emulation. Recommendation: reserve `mm8`; keep `dot4` out, or in only as `dot4_u8` behind `mm8`.
## 3. The genesis reserve entry (spec text for 1.13.2)
Proposed wording, to go under 1.13.2 as the first named reserve family once the conformance runs of section 5 pass:
> Reserve family R1, `mm8` (integer matrix). Semantics: section 2.2 of `docs/analysis/int8-matrix-family.md`,
> uint8 operands from `src` and `src2` in the m8n8k16 fragment layout, one int32 element of C per lane selected by
> the immediate `bit`, added into `dst` modulo 2^32. Weight at unlock `W_new = 4` points, taken proportionally from the
> ten live non-load families (the load weight and count are untouched, 1.13.1). Edge vectors, each a hand-built unit
> run on every vendor: all bytes 0xFF in A and B (C = 16 x 65,025 = 1,040,400 everywhere); all bytes 0x80 (C = 16 x
> 16,384 = 262,144); A all zero (C = 0); `dst` = 0xFFFFFFFF with a nonzero C (the wrap); alternating 0x00 and 0xFF by
> lane (the fragment mapping: C[r][c] nonzero only where the row and column bytes meet); `bit` = 0 and 1 on the same
> fragments. Unlock: at the start of era n = 4 (two years after genesis, DAA 62,208,000), or earlier by the 90%
> signalling path of section 5.7; never by a release.
The era-4 choice is deliberate: two years is long enough for the three vendors' tensor paths (and Apple's Metal 4
feature-set coverage) to be in every miner's driver, and short enough to land before any chip built against the
launch instruction set has paid back (approximate; a chip programme is 12 to 24 months, approximate, from memory).
Reserve rule for a vendor that can only emulate. Spec 1.13.2 as written requires conformance on every vendor; it says
nothing about cost. Proposed addition:
> A family enters the reserve when it is bit-exact on every vendor of 1.15. A vendor that reaches the result only by
> emulation (no instruction or library path) does not block entry if the measured penalty of the emulation on that
> vendor, on the family's own probe (a dependent chain of the op, G ops/s against the same vendor's integer ALU chain),
> is at most 8x per op, AND the family's weight at unlock keeps the emulating vendor's hash-rate loss under 5% on the
> memory-hard hash (the hash is latency-bound, so a per-op penalty on 4% of the instructions is a small fraction of a
> hash whose time is 128 dependent DRAM reads; the 5% is checked on the vendor's card with the family live, not
> computed). A family whose emulation exceeds either bound stays out of the reserve until the vendor ships a path.
With tonight's numbers: on Apple the unsigned `dot4` emulation is 1.6x per op (inside the bound); the signed one 4.7x
(inside, but why pay it); `mm8` through Metal 4's matmul2d is a path, not an emulation, and its cost is owed.
## 4. dp4a-class throughput, measured so far
Probe: a dependent chain of one dot4 per step per lane (`acc = dot4(x, y, acc); x = x * K + acc; y = rotl(y, 7) ^
(acc + s)`), 1,048,576 lanes x 4,096 steps, best of 3, device time, bit-exact against a CPU reference on two lanes
per run, beside the ALU chain of the 9070 XT bench-log entry (`x = x * K + rotl(y, 7); y = (y ^ x) + s`, 5 ops per
step counted). Sources: `proto-metal/dot4-probe.swift` (Metal), `proto-opencl/dot4-probe.c` (OpenCL: scalar, the
`cl_khr_integer_dot_product` `dot`, AMD `__builtin_amdgcn_sudot4`, NVIDIA inline PTX `dp4a.s32.s32`),
`proto-cuda/dot4-probe.cu` (CUDA `__dp4a` and the scalar emulation, for a PC with nvcc). Each OpenCL variant is built on
its own and a variant the platform cannot compile prints a "build failed" row.
| Card, API | Date, command | ALU chain, G steps/s | dot4 signed emulation, G dot4/s | dot4 unsigned emulation, G dot4/s | dot4 intrinsic, G dot4/s | Penalty of the emulation per op (ALU steps per dot4) | ok (bit-exact) |
|---|---|---|---|---|---|---|---|
| Apple M5 Max, Metal | 5 October 2026 20:0x UTC, `with-lock.sh measure ./dot4-probe` (swiftc -O), GPU start-to-end time | 879.8 (4.882 ms) | 188.2 (22.82 ms) | 548.2 (7.834 ms) | none exists | signed 4.7x, unsigned 1.6x | yes, all three kernels |
| Apple M5 Max, Apple OpenCL 1.2 | same, `with-lock.sh measure ./dot4-probe-cl --device 0`, event time | 871.5 (4.928 ms) | 188.4 (22.80 ms) | not in this probe | `cl_khr_integer_dot_product` not listed; the kernel using `dot(char4, char4)` compiled anyway and ran at 846 G/s but MISMATCHED the CPU reference on every lane checked (Apple's `dot` on char4 is not an integer dot; the extension macro must gate it) | signed 4.6x | alu and dot4e yes; dot4_khr NO |
| RTX 5090 (PC 1, ae432dc7), NVIDIA OpenCL 3.0 CUDA, driver 617.14 | 5 October 2026 20:29 UTC, job `run-dot4-20261005` (`relay/playbooks/dot4-probe.ps1`, both mining cards switched off in the app first, restored after; `node tools/jobs.mjs run-dot4-20261005`), event time | 8,753.5 (0.491 ms) | 1,239.1 (3.466 ms) | not in the OpenCL probe | 7,453.6 (0.576 ms) via inline PTX `dp4a.s32.s32` | emulation 7.1x; the intrinsic 1.17x (the chain is one dp4a plus 3 ops against 5 ops), so the emulation costs 6.0x the instruction | yes, all three |
| RX 9070 XT (PC 1, gfx1201, eGPU), AMD OpenCL 2.0 AMD-APP 3683.0 (PAL,LC) | same job, same time | 701.4 (6.124 ms) | 480.8 (8.932 ms) | not in the OpenCL probe | 664.3 (6.465 ms) via `__builtin_amdgcn_sudot4(true, a, true, b, acc, false)`: the Adrenalin OpenCL C compiler accepts the clang builtin and emits `v_dot4_i32_iu8` | emulation 1.46x; the intrinsic 1.06x, so the emulation costs 1.38x the instruction | yes, all three; the older 3652.0 platform entry for the same card gave 696.2 / 501.7 / 683.6 |
| Ryzen 9800X3D gfx1036 (PC 1, integrated RDNA 2, 2 CUs) | same job | 40.6 (105.9 ms) | 15.8 (272.3 ms) | | `sudot4` does not build: "needs target feature dot8-insts" (RDNA 2 has `dot1-insts`' `v_dot4_i32_i8`, the `sdot4` builtin, which the probe did not try) | emulation 2.6x | alu and dot4e yes |
| Every PC device | `cl_khr_integer_dot_product` not listed on NVIDIA (OpenCL 3.0) or AMD (2.0); the pragma draws "unknown OpenCL extension" on both and the `dot(char4, char4)` kernel does not build | | | | | | |
Reading across the three cards. Per dot4 at the hardware rate: the 5090 does 7.45 T dot4/s (one `dp4a` per step, 0.85
of its ALU-chain step rate), the 9070 XT 0.66 T (0.95 of its ALU-chain rate), the M5 Max 0.55 T at best (the unsigned
emulation; no instruction). On the ALU chain the 5090 is 12.5x the 9070 XT and 10x the M5 Max; on hardware dot4 it is
11.2x the 9070 XT, so the family does not widen the AMD gap, and 13.6x the M5 Max, so Apple's emulation widens its gap
by 1.4x on this op (approximate: one probe shape, the ratios of best-of-3 numbers). The signed emulation is where the
vendors differ most: 7.1x the ALU step on NVIDIA, 4.7x on Apple, 1.46x on AMD (AMD's compiler and byte-permute
hardware make the four sign-extended products nearly free; the NVIDIA OpenCL compiler does not pattern-match the
emulation into `dp4a`, which the 6x gap between `dot4e` and `dot4_nv` shows). None of this is a hash-rate number: the
hash is bound by 128 dependent DRAM reads, and a family at W_new = 4 adds about 21 of these ops per hash per lane
against about 1.2 microseconds of memory latency per hash per lane (approximate), so the per-op penalties above turn
into hash-rate losses well under 5% on every card, to be measured with the family live.
Reading of the Mac numbers. The ALU chain's 880 G steps/s on the M5 Max is the integer baseline (5 ops per step
counted, so about 4.4 T int ops/s, approximate; the 5090's 8,754 G steps/s is about 43.8 T, against the whitepaper's
104.8 peak INT32 TOPS which counts a multiply-add as two). A signed dot4 emulated as `int4(as_type<char4>(a))` products costs
4.7 of those steps; the unsigned form 1.6 steps. The 3x gap between the two is the sign extension (Metal lowers the
unsigned byte extraction to masks that fold into the multiplies, approximate reading of the result, not of the
compiled code). Both are far under the 8x bound of section 3, and the hash spends its time on DRAM reads, so a per-lane
`dot4` family would cost Apple a few percent at W_new = 4 (to be measured with the family live, not computed). The
Apple OpenCL `dot(char4, char4)` mismatch is the kind of thing the edge vectors of section 3 exist to catch.
## 5. What is owed or unverified
| Item | State |
|---|---|
| dp4a throughput on the RTX 5090 through NVIDIA OpenCL inline PTX | measured (section 4); the CUDA `__dp4a` form (`proto-cuda/dot4-probe.cu`) is unrun (no nvcc job tonight) and is a cross-check, not a gap |
| `sudot4` on the 9070 XT through Adrenalin's OpenCL C; `cl_khr_integer_dot_product` on the 3683.0 platform | measured: the builtin works and emits the instruction; the extension is not listed and the `dot(char4, char4)` kernel does not build |
| `sdot4` (`dot1-insts`) on RDNA 2 (gfx1036) | not tried; the probe only carries `sudot4` |
| Metal 4 `matmul2d` uchar x uchar into int on the M5 Max: wrap or saturate at the int32 edge, native or emulated, throughput | owed (a second Metal probe; the API needs a tensor set-up the dot4 probe does not have) |
| Metal Feature Set Tables: which Apple GPU families run int8 matmul2d natively | not read tonight |
| RDNA 3 and RDNA 4 ISA guides: the instruction text itself (names taken from LLVM and GPUOpen) | AMD's CDN refused the downloads tonight |
| `mm8` on AMD: the exact byte permutation between the PTX m8n8k16 fragment layout and the RDNA WMMA 16x16x16 layout | design, to be written with the kernel |
| The hash-rate cost of the family live at W_new = 4 on each vendor (the 5% rule of section 3) | owed, needs the generator change (not tonight) |
| Edge vectors of section 3 as files | owed, with the generator change |

View file

@ -0,0 +1,411 @@
# Proving methods: why the prover needs 14 GB, what else exists, and how a 12 GB card gets to prove
5 October 2026, from the project lead at 22:05 UTC: "if this doesn't enable 12 GB cards, then do a full deep research task on proving
and see if there are different methods." "This" is the prover-floor agent's patch of SP1's GPU server (branch
`prover-floor`), running tonight. This document is research and reading, not measurement: every number of ours is from
`docs/bench-log.md` with its entry named; every claim about another system cites its repository file, its documentation
page or its paper, or is labelled approximate. Status words follow `docs/spec/00-overview.md` 0.2. Day estimates follow
the project lead's rule of 3 October 2026: hours of agent time, never weeks.
The facts this starts from (bench-log, "proving v1", 5 October 2026; `docs/plans/proving-v1.md`; `docs/analysis/amd-proving.md`):
| Fact | Number |
|---|---|
| SP1 6.8.1's GPU server, an empty shard (280,706 cycles), the card to itself | 13,874 MiB peak, 2.2 s compressed |
| The adopted v1 shard (`S_p` 30,000 pgas, 4,717,439 cycles) | 20,434 MiB, 4.3 s; 22,210 MiB and 13.2 s beside the miner |
| The prototype shard (6.75 M pgas, 60.4 M cycles) | 28,307 MiB, 10.8 s; flat at 28.3 GB from 20 M cycles up |
| Aggregation, chained, per block, on a mining 5090 | 9.6 to 9.7 s; 2.2 to 2.5 s with the card to itself |
| The CPU path (PC 1, 16 cores) | 282 s a shard whatever its size, 29.5 to 30.5 GB RSS |
| AMD and Apple GPUs | no zkVM proves on AMD; RISC Zero has a Metal prover, SP1 does not |
| The promise | `site/litepaper.html`: "Target: shard size will be set so a 12 GB card proves one shard in about 20 seconds"; the design goal is every block proven within about a minute by the miners' own cards |
## 1. The memory anatomy of a STARK-based zkVM prover, and why the floor is where it is
### 1.1 What SP1 6.8.1 is
SP1 6.x is not the univariate FRI STARK of the earlier SP1 releases (Succinct calls Hypercube the "first zkVM built entirely on a multilinear polynomial-based proof system", blog.succinct.xyz, sp1-hypercube, 20 May 2025). The crates it pulls say what it is: `slop-multilinear`, `slop-sumcheck`,
`slop-jagged`, `slop-stacked`, `slop-basefold`, `slop-whir` (the `~/.cargo/registry` of this Mac; `proving/igneum-prove/Cargo.toml`
pins `sp1-sdk = "=6.8.1"`). The architecture Succinct calls Hypercube: the execution trace is a set of multilinear
polynomials over the 31-bit KoalaBear field (`sp1-hypercube-6.8.1/src/verifier/config.rs`: `SP1BasefoldConfig =
Poseidon2KoalaBear16BasefoldConfig`), the constraints are checked by a zerocheck sumcheck and the lookups by a LogUp GKR
(`sp1-gpu/crates/zerocheck`, `sp1-gpu/crates/logup_gkr`), and the polynomial commitment is "jagged": every table's
columns, whatever their heights, are concatenated into one long vector, stacked into rows of height `2^log_stacking_height`
and committed with BaseFold, a FRI-like folding over a Reed-Solomon code (`sp1-gpu/crates/basefold/src/fri.rs`,
`slop-basefold-6.8.1/src/verifier.rs`). The proof system parameters, from `sp1-primitives-6.8.1/src/fri_params.rs` and
`sp1-prover-6.8.1/src/components.rs`:
| Parameter | Value | Where |
|---|---|---|
| Field | KoalaBear, 31 bits, 4 bytes an element; extension degree 4 (16 bytes) | `sp1-primitives` |
| Core stage: Reed-Solomon blowup | `CORE_LOG_BLOWUP = 2`, so the codeword is 4x the data | `fri_params.rs:5` |
| Core stage: stacking height, maximum rows per table | `CORE_LOG_STACKING_HEIGHT = 21`, `CORE_MAX_LOG_ROW_COUNT = 22` | `components.rs:16,17` |
| Core shard limits (the executor's cut) | `MAX_SHARD_SIZE = 2^24` cycles, `ELEMENT_THRESHOLD = 2^28 + 2^27 = 402,653,184` trace elements, `HEIGHT_THRESHOLD = 2^22` rows | `sp1-core-executor-6.8.1/src/opts.rs:9-12` |
| Recursion (compress) stage | blowup 2 (`RECURSION_LOG_BLOWUP = 2`), stacking height 20, max rows 2^21 | `fri_params.rs:6`, `sp1-verifier-6.8.1/src/compressed/config.rs:1,2` |
| Shrink and wrap stages | blowup 3 (8x), 22 bits of grinding, stacking 18 and 21 | `fri_params.rs:17,18,7`, `components.rs:37-40` |
| Recursion arity | 4 proofs per compose step (`DEFAULT_MAX_COMPOSE_ARITY = 4`, `DEFAULT_MAX_REDUCE_ARITY = 4`) | `sp1-prover-6.8.1/src/worker/config.rs:183,193` |
| Workers | 4 core workers, 8 recursion prover workers, 4 recursion executors, 4 deferred workers, buffers of 4 to 8 | `worker/config.rs:188-205` |
The stages a shard goes through (`sp1-prover-6.8.1/src/worker/controller/*.rs`): execute (the RISC-V executor cuts the
run into core shards at the thresholds above); core (one jagged-PCS proof per core shard, on the GPU); normalize and
compose (each core proof is verified inside a recursion program, then proofs are folded 4 at a time until one remains,
the "compressed" proof, 1,272,897 bytes for every shard we have proven, bench-log 4 and 5 October); deferred (what the
aggregator uses: `verify_sp1_proof` inside a guest, `proving/igneum-prove/aggregator/src/main.rs`); shrink and wrap
(to a BN254 STARK, then Groth16 or Plonk; not run here, ledger P3).
### 1.2 The terms, and which scale with the shard
Every STARK-family prover holds these buffers on the device at its peak, in some order and with some overlap. The
sizes below are from the constants of 1.1 and the allocation code of `sp1-gpu`; where a buffer's size is the actual
trace rather than the maximum, the row says so.
| Term | What it is | Size rule | SP1 6.8.1 on a 32 GB card | Scales with the shard? |
|---|---|---|---|---|
| Main trace | the witness: one element per cell of every table the shard touched | `cells x 4 bytes`, where cells = sum over tables of rows x columns; the executor cuts a new core shard at 402,653,184 cells | the device buffer is allocated at the MAXIMUM, not the actual trace: `allocate_and_initialize_traces` takes `max_trace_size` and allocates `max_trace_size` felts plus `max_trace_size / 2` u32 of index (`sp1-gpu/crates/jagged_tracegen/src/lib.rs:484-503`), 6 bytes a cell; the core prover's `max_trace_size` is `element_threshold + 2^21` (`prover_components/src/builder.rs:70-71`): **2.26 GiB** on a card over 30 GB, 1.61 GiB on a 24 GB card (the threshold drops by 2^26 + 2^25 + 2^24 when memory is 30 GB or under, `builder.rs:41-45`) | no: fixed at the maximum shard, whatever the trace |
| Preprocessed trace | the program's own tables (the ELF as a `Program` AIR, the byte and range tables) | program size x its columns plus 2 x 2^16-class tables | small for a 2.8 MB guest ELF (`elf/manifest.json`); not isolated | with the guest, not the shard |
| Codeword (the LDE) | the stacked polynomial encoded at rate 1/4 for BaseFold | `stacked cells x 4 (blowup) x 4 bytes`; the stacked length is the actual cell count padded to a multiple of 2^21 | 4.7 M cycles: approximate, the actual trace; 60 M cycles: the shard is 7 to 15 core shards of up to 402 M cells, each encoded to 6 GiB at the blowup, one or more in flight | yes, up to the core-shard cap; past the cap the shard count grows and the per-shard term stays |
| Merkle commitment | Poseidon2 hashes of the codeword rows | `rows x 8 elements x 4 bytes x 2`, rows = 2^21 x blowup | about 0.5 GiB at full stacking, approximate | with the stacked rows |
| Zerocheck and GKR | the constraint sumcheck over the extension field, and the LogUp GKR layers | extension elements are 16 bytes; the sumcheck holds a folded copy of the trace in the extension field, which is 4x the base trace at the first round and halves each round | up to about 4x the live trace in the first round, approximate (`sp1-gpu/crates/zerocheck/src/primitives.rs:174,287`: `Buffer<Ext>` of `new_total_length`) | yes |
| Recursion traces | the normalize and compose programs' own traces, verifying core proofs | fixed-shape programs: `RECURSION_TRACE_ALLOCATION = 2^27` cells (`builder.rs:15`), allocated at 6 bytes a cell: **0.75 GiB** per recursion prove, at blowup 4 a 2 GiB codeword plus its own zerocheck | fixed per recursion step; the number of steps is log4 of the core-shard count | no (per step) |
| Shrink and wrap traces | the two last stages, not run by us | 2^25 and 85,376,340 cells (`builder.rs:16,19`): 0.19 and 0.48 GiB | only when wrapping | no |
| Proving keys and program cache | the recursion programs (`vk_map.bin`, the normalize cache of 5 programs) and the shard program's setup | `DEFAULT_NORMALIZE_PROGRAM_CACHE_SIZE = 5` (`worker/config.rs:192`); the key setup took 14.6 s on the 5090 (bench-log 4 October) | not isolated | no |
| Pinned host buffers | the staging copies on the PC side | 4 core workers x `max_trace_size` x 4 bytes = **6.0 GiB** of pinned RAM, plus 4 x 0.5 GiB for recursion (`prover_components/src/components.rs:99-103`, `builder.rs:76,95`) | this is the 7.9 GB WSL2 working set measured on 5 October | no |
| The allocator | CUDA's default memory pool with its release threshold set to `u64::MAX` (`sp1-gpu/crates/cuda/src/task.rs:152,190-199`): freed blocks are never returned to the driver | `nvidia-smi` therefore reports the high-water mark of everything above, and it stays until the server exits | this is why the memory curve is flat between shards of different size | no |
Two facts from this table explain the measurements:
1. **The server refuses small cards by code.** `local_gpu_opts()` reads the card's total memory, adds 4 GB, and panics
under 24: `"Unsupported GPU memory: {gpu_memory_gb}, must be at least 24GB"` (`sp1-gpu/crates/prover_components/src/builder.rs:35-38`).
A 16 GB card (16 + 4 = 20) and a 12 GB card (16) never start; a 20 GB card is the smallest that does. The 13.9 GB
floor measured on the 5090 is therefore not the whole story for a 12 GB card: on this build the card is refused
before any buffer is allocated. Any route through SP1's GPU server starts by removing this line.
2. **The environment knobs do not reach the floor** because the same function overwrites `element_threshold` with the
compile-time constant (`builder.rs:41-48`); only `HEIGHT_THRESHOLD` passes through, which is why the sweep's
`ELEMENT_THRESHOLD 2^26` rows changed nothing and `HEIGHT_THRESHOLD 2^20` took 5.4 GB off the 60 M-cycle shard
(bench-log, "the 12 GB requirement", 5 October 2026). The worker counts only slow the proof (11.4 s to 20.8 s)
because the device buffers are sized by `max_trace_size`, not by the worker count.
### 1.3 Why the floor is 13.9 GB for an empty shard
With the release threshold at `u64::MAX`, the peak is the high-water mark over the whole pipeline. For an empty shard
the core trace is small (280,706 cycles), so the fixed-shape terms dominate: the 2.26 GiB main-trace buffer allocated
at the maximum, the recursion step over a fixed-shape normalize program (a 0.75 GiB trace buffer, its 4x codeword in
the extension field for the zerocheck, its Merkle tree), the proving-key and program caches, and the deferred and
compose machinery that a compressed proof always runs once. The decomposition of the 13.9 GB into those terms is
approximate until the prover-floor agent's profile lands (branch `prover-floor`, tonight): the figure that is not
approximate is that none of it is the witness (5 to 22 KB a shard) and none of it is the shard's cycles (the same
13.9 GB at 280 k cycles and 556 k cycles, bench-log "the S_p curve").
The step from 13.9 GB (empty) to 20.4 GB (4.7 M cycles) is the live trace: the 4.7 M-cycle shard is one core shard
(its trace area is under 402 M cells, so it was not split; approximate from the memory curve, the cell count is not
logged by the host), and its codeword, zerocheck and GKR buffers are sized by its actual cells. The step from 20.4 GB
to 28.3 GB (20 M cycles and up) is the second and later core shards in flight at once: 4 core workers with a buffer of
4 (`worker/config.rs:188,189`) let several core shards' codewords exist at the same time; past 20 M cycles the pipeline
is full and the peak is flat, which is what the curve shows (28,371 MiB at 20 M cycles, 28,307 at 40 M and 60 M).
### 1.4 The theoretical floor for our guest at the adopted shard
If every buffer were sized to the shard rather than to the maximum, the adopted v1 shard (4.7 M cycles) would need,
approximate, from the rules of 1.2:
| Term | Rule | Approximate bytes |
|---|---|---|
| Main trace, actual | 4.7 M cycles x about 60 cells a cycle (the `Add` and `Addi` tables cost 33 and 30 columns a row, a memory access adds 20 and a global interaction 241: `sp1-core-executor-6.8.1/src/artifacts/rv64im_costs.json`) | about 280 M cells, 1.1 GB |
| Codeword at blowup 4 | 4x | 4.5 GB |
| Zerocheck first round in the extension field | 4x base, halving each round | 4.5 GB at the peak round, falling |
| Merkle tree | rows x 32 bytes x 2 | 0.3 GB |
| Recursion step, fixed | 2^27 cells x 6 bytes plus its 4x codeword and extension copies | 2 to 3 GB, approximate |
| Keys and caches | | under 1 GB, approximate |
| Peak, if the core stage and the recursion stage do not overlap and the pool releases | | **about 10 to 11 GB**; about 6 GB if the blowup-4 codeword is replaced by a rate the sumcheck does not need (see 2.4) |
So the adopted shard is, on paper, a 12 GB card's shard with the server re-sized and nothing else changed, and it is a
12 GB card's shard with 4 GB to spare if the shard is halved (`S_p` 15,000 pgas, 2.4 M cycles: the planner cuts at
transaction boundaries to any budget, `core/src/plan.rs`, and the fee switch of 5 October already moved `S_p` once).
What the paper figure does not say is the time: a smaller card proves slower, and 60 s with the miner running is the
bound (section 3). The prover-floor agent is measuring the real figure; this section says what it should find and why.
### 1.5 RISC Zero's anatomy, for comparison
RISC Zero is the FRI STARK the textbooks describe, and its constants make the same table easy to read
(`~/.cargo/registry`, `risc0-zkp-3.0.4/src/lib.rs`, `risc0-circuit-rv32im-4.0.4/src/zirgen/defs.rs.inc`):
| Term | Value | Where |
|---|---|---|
| Field | BabyBear, 31 bits; extension degree 4 | `risc0-core` |
| Segment size | `DEFAULT_SEGMENT_LIMIT_PO2 = 20` (1,048,576 cycles), `MIN_CYCLES_PO2 = 13`, `MAX_CYCLES_PO2 = 24`; `DEFAULT_MAX_PO2 = 22` for the verifier | `risc0-circuit-rv32im-4.0.4/src/execute/mod.rs:39`, `risc0-zkp-3.0.4/src/lib.rs:35-38`, `risc0-zkvm-3.0.4/src/receipt.rs:884` |
| Trace width | data 211 + accum 103 + code 1 = 315 columns; globals 90, mix 36 | `defs.rs.inc:7-11` |
| Blowup | `INV_RATE = 4`; 50 queries; FRI fold 16 | `risc0-zkp-3.0.4/src/lib.rs:41-51` |
| Recursion | lift, join and resolve programs at `RECURSION_PO2 = 18` rows | `risc0-zkvm-3.0.4/src/host/recursion/prove/mod.rs:58` |
| The GPU buffers | `check + ctrl + data + accum + mix + out` elements x 4 bytes, printed by the CUDA HAL at `eval_check` | `risc0-circuit-rv32im-4.0.4/src/prove/hal/cuda.rs:181-200` |
From those constants the trace of a segment is `2^po2 x 315 x 4` bytes and its LDE 4x that, so, approximate: po2 18 is
0.3 GB of trace and 1.5 GB with the LDE, po2 19 is 3.1 GB, po2 20 is 6.2 GB, po2 21 is 12.3 GB, before the check
polynomial, the extension-field accumulators and the Merkle trees. Two things follow. A RISC Zero segment at the
default 2^20 is in the same class as one SP1 core shard, not smaller. And a RISC Zero segment at 2^18 or 2^19 is a
2 to 4 GB object: the only reason a 12 GB card could not prove one is the fixed overhead of the recursion circuits
(2^18 rows each) and the allocator, which is the measurement the prover-floor agent takes on PC 2 if SP1 cannot go
under 11 GB. Section 2.2 carries the documented numbers.
### 1.6 Which terms the shard size can move, and which it cannot
| Lever | Moves | Does not move |
|---|---|---|
| Our `S_p` (pgas per shard) | the live trace, the codeword, the zerocheck: everything in 1.2 marked "yes" | the maximum-sized buffers, the recursion step, the keys, the pool |
| SP1's `HEIGHT_THRESHOLD` (the one knob the server honours) | the rows per table in one core shard, so the live buffers | the fixed terms (measured: 13,861 MiB on the empty shard with every knob at its minimum) |
| A server patch: size `max_trace_size` to the shard, release the pool, one core worker | the 2.26 GiB buffer, the high-water mark, the in-flight count | the recursion step's fixed shape and the key caches |
| A different proof system | the blowup (sumcheck-only and linear-code systems have none, 2.4), the recursion shape | the trace itself: a RISC-V cycle costs tens of cells in every zkVM |
## 2. Every current proving route
Read 5 October 2026, 22:10 to 23:00 UTC, by four research agents and this one; every cell names its page or file.
"Not documented" means the project publishes no figure, which for a memory floor is itself the finding.
### 2.1 The zkVMs with a GPU prover
| Prover | Proof system, field, chunk | GPU support and the documented minimum memory | Throughput, on what | Verification of the recursive proof; wrapper | Licence | State, October 2026 |
|---|---|---|---|---|---|---|
| **SP1 6.8.1** (ours) | Hypercube: multilinear, jagged PCS, BaseFold, LogUp GKR; KoalaBear; core shards of up to 2^24 cycles and 402 M cells (section 1) | CUDA only. Docs: "24GB or more VRAM", compute capability 8.0+, Linux x86_64 (docs.succinct.xyz, hardware-acceleration page). Code: panic under 20 GB physical (`builder.rs:35-38`). Measured here: 13.9 GB floor, 20.4 GB at the adopted shard, 28.3 GB at the prototype shard. Issue #2950: two clients on a 48 GB L40S hold 41 to 43 GB; a single 6 GiB tensor allocation failed | 4.3 s for the adopted shard, 10.8 s for the prototype one on a 5090 (bench-log); Succinct: 99.7% of Ethereum blocks under 12 s on 16 x RTX 5090 (blog.succinct.xyz, 18 Nov 2025) | compressed proof 1,272,897 bytes, verified in 0.032 to 0.040 s here (`--mode verify-segment`); Groth16 about 260 bytes and about 270 k gas, Plonk about 868 bytes and 300 k gas (docs, proof-types page); the Groth16 wrap needs about 14 GB of host RAM, Plonk about 60 GB (hardware-requirements page) | Apache-2.0 or MIT for the repository including `sp1-gpu/` (`LICENSE-APACHE`, `LICENSE-MIT` at the root; no separate licence under `sp1-gpu/`); `sp1-cluster` is Business Source 1.1 | v6.8.1 of 24 Sep 2026 is the latest tag; mainnet for Ethereum proving since 19 Feb 2026 (blog); AMD port PR #2668 closed unmerged 20 Mar 2026; no Metal, Vulkan or WebGPU |
| **RISC Zero 3.0.x** | FRI STARK (DEEP-ALI), BabyBear, Poseidon2, blowup 4, 50 queries; segments of `2^po2` cycles, default po2 20, allowed 13 to 24 (section 1.5); lift, join, resolve recursion at 2^18 rows; keccak as a separate circuit | CUDA and **Metal** (`risc0/sys/kernels/zkp/metal/`; on Apple silicon the Metal path is on automatically, `risc0/zkvm/build.rs`). Documented memory per segment: Bento design page, 1 M cycles 9 to 10 GB, 2 M 17 to 18 GB, 4 M 32 to 34 GB; Boundless performance page, the largest `SEGMENT_SIZE` per card: 8 GB card po2 19, 16 GB po2 20, 20 GB po2 21, 40 GB po2 22; docs: "less than 10 GB available: change the segment size limit" (dev.risczero.com, local proving); `env.rs:190-192`: "lowering this value by 1 will cut memory consumption by about half". PR #3761 (June 2026): po2 22 did not fit a 24 GB 4090 until the `low_vram` buffer reuse | 4090: 808 kHz at po2 21, 1,207 kHz at po2 22 with PR #3761 (end to end to a succinct receipt); Apple M2 Pro about 14 kHz on the 2023 datasheet, approximate (the page was unreachable tonight); real-time Ethereum on about 160 x 4090 (blog, approximate) | succinct receipt 222,668 bytes, constant; about 100 ms to verify, approximate (`gsr-stark-verifier` PR #5, mirrored docs); Groth16 seal 256 bytes, about 200 to 300 k gas, approximate; the Groth16 wrapper is x86 only, not on Apple silicon (docs) | Apache-2.0 or MIT, CUDA and Metal kernels included (`risc0/sys/kernels/zkp/cuda/eltwise.cu:1-13`); Bento is BSL 1.1 with a change date already passed | v3.0.6 of 17 Jul 2026 on the maintained line; `main` is 5.0.0 with no release body; `RISC0_PROVER=actor` multi-GPU scheduler experimental since 3.0.1 (`r0vm/src/actors/factory.rs:183-195` carries measured per-po2 memory tokens: po2 18 = 8, 19 = 10, 20 = 15, 21 = 24; lift and join = 3) |
| **Airbender** (Matter Labs) | DEEP STARK, FRI, Mersenne31; chunks of 2^22 cycles; Boojum then FFLONK wrap (docs.zksync.io, airbender page) | CUDA only. "any GPU with 22GB RAM" for production (zksync.io/airbender); the code has memory presets `GiB21` (24 GB cards) and `GiB30`, raised from 29 because Ethereum blocks failed to allocate at 29 GiB (PR #448, `gpu/execution_prover/src/prover/config.rs`); the final SNARK is CPU with about 150 GB of RAM (`docs/gpu.md`) | one H100: 21.8 MHz base layer, 8.5 MHz end to end, about 35 s an Ethereum block (June 2025 post); ethproofs.org today: 4 x 5090 2.3 s average | FFLONK over BN254 on chain; gas not published | MIT or Apache-2.0 | v0.6.0-rc.2; Veridise audit Feb to Apr 2026; live for ZKsync Atlas chains |
| **ZisK** (Polygon spin-out) | eSTARK over Goldilocks (pil2-stark), Poseidon2, approximate; main instance 2^22 to 2^23 rows, chunks of up to 2^22 steps (PR #1238) | CUDA only, CUDA 12.9+; **no VRAM floor documented**; workers need about 32 GB of host RAM, the assembly emulator 64 GB (docs, limits and distributed pages); Cysic's Venus fork submits from one RTX 4090 | 4 x 5090: p99 9.62 s on Ethereum blocks (Aug 2026); 24 x 5090 6.56 s average (Nov 2025) | PLONK wrapper verified by Solidity (`zisk-contracts`); 128-bit claimed | Apache-2.0 or MIT | v1.3.1-alpha, 30 Sep 2026, "undergoing security and correctness audits" (README) |
| **OpenVM 2.0** (Axiom) | SWIRL: sumcheck, zerocheck, LogUp GKR, stacked reduction into WHIR; BabyBear; segments by metered trace height (blog.openvm.dev/2.0) | CUDA only; "at least 24GB of VRAM": L40, 4090, L40S, 5090 (blog.openvm.dev/openvm-gpu) | 11.4 MHz on one 5090, 139 MHz on 16; 2.1 preview: 4 x 5090 p99 9.7 s | STARK proof under 300 KB; Halo2-KZG wrapper, 316 k gas; the Halo2 wrap 8.1 s on a 5090 | MIT or Apache-2.0, GPU prover included | v2.0.2 of 14 Aug 2026; zkSecurity audit of SWIRL; Scroll's prover builds on it |
| **Pico** (Brevis) | Plonky3 STARK, KoalaBear default; chunk size a parameter with no documented default | CUDA via `pico-gpu`; **no VRAM figure published**; every run on 32 GB 5090s | Prism 2.1: 16 x 5090 over two machines, 4.87 s average on Ethereum blocks | Groth16 via gnark | core MIT or Apache; **`pico-gpu` is BUSL-1.1** and "not recommended for production" (its README) | v2.1.2, Aug 2026; Sherlock audit |
| **Ziren** (ZKM, MIPS) | Plonky3-class, KoalaBear, LogUp GKR, WHIR | CUDA 12, compute capability 8.6+, "24 GB VRAM or higher"; the GPU prover is a Docker image pinned by digest, source "planned H1 2026" (docs.zkm.io prover page; an independent evaluation of v1.1.4 says the GPU path is not open) | one 5090: 5.9 MHz on a 288 M-cycle block; 4 GPUs 3.1 to 3.3x | compressed proof 603 KiB; Groth16 or PLONK | core MIT or Apache; GPU image licence unspecified | v1.2.7; no public audit cited |
| **Stwo / S-two** (StarkWare) | Circle STARK over Mersenne31; blowup 1 (rate 1/2), 70 queries, 26 bits of grinding | **CPU SIMD first** (AVX2, AVX-512, NEON, WASM); GPU through ICICLE-Stwo (Ingonyama): about 3 GB of trace in GPU memory, out of memory from 2^23 rows; a WebGPU port of the constraint evaluation (zkSecurity blog, April 2025); client-side proving under 1 GB after a spill allocator (third-party PR) | 620 k Poseidon2 a second on an M3 laptop | via a Cairo verifier (proofs of proofs); sizes not published here | Apache-2.0 | live on Starknet mainnet since 3 Nov 2025; no RISC-V guest of its own (Nexus 3.0 is the RISC-V zkVM on it, BUSL-1.1 until 2029) |
| **Jolt** (a16z) | sumcheck and lookups (Lasso lineage, Twist and Shout memory checking); PCS Dory over BN254 by default, or **Akita**, a lattice commitment over a 128-bit prime field (Sep 2026, "Lattice Jolt"); RV64IMAC; no continuations (the book's recursion page is "under construction") | **No CUDA in the public repository** (LayerZero's "Jolt Pro" CUDA port is private); **Metal**: PR #1938 merged 30 Sep 2026 (the `jolt-metal` runtime crate), PR #1733 (the full prover on Metal) still a draft. Memory: "about 200 bytes per cycle" with Akita (a16z substack, Sep 2026); the book: "under 2 GB of memory per million cycles"; a streaming prover bounded to "a few GBs" is planned, not shipped | over 2 M cycles a second on a laptop CPU with Akita, over 10 M with Metal on a Apple laptop (a16z substack, Sep 2026); PR #1733: M5 Max, 2^25 cycles in 19.8 s, 2^27 in 77 s at an 89.4 GiB footprint | proof about 50 KB (Dory) or 65 to 80 KB (Akita); verify sub-second, approximate; on-chain 1.3 to 2 M gas estimated in 2024; no Groth16 wrapper shipped | MIT or Apache-2.0 | `v0.3.0-alpha` is the last tag (1 Oct 2025); README: "not suitable for production use"; no audit |
| **Ceno** (Scroll) | GKR tower prover, BabyBear, WHIR or BaseFold PCS; RV32IM | CUDA, but the real HAL is in a **private** `ceno-gpu` repository (the public one is a mock); no memory numbers | 2 GPUs 1.6x over one (PR #1403, Sep 2026) | via OpenVM recursion to Halo2 | Apache-2.0 | README: "under construction and not suitable for use in production" |
| **Nexus 3.0** | on Stwo (Circle STARK, M31) | no GPU path documented | none published | not published | **BUSL-1.1** until 10 Feb 2029 | last push 6 Jan 2026; folding (Nova family) abandoned June 2025 for the STARK |
| **Valida** (Lita) | Plonky3 STARK | no GPU; a CUDA port "underway" in July 2025 | none current | not published | Apache or MIT | dormant since Sep 2025; documented soundness issues in its own benchmarks page |
| **Powdr** | no longer a zkVM: `powdrVM` archived; powdr is autoprecompiles on OpenVM | OpenVM's | OpenVM's | OpenVM's | MIT or Apache | tooling layer; "DO NOT USE FOR PRODUCTION" |
| **Binius / Binius64** (Irreducible) | binary-field SNARK, BaseFold-style FRI over GF(2^64) words | CPU SIMD only; the FPGA work was dropped 9 Sep 2025 ("FPGAs underperformed GPUs"); no GPU | ECDSA aggregation about 5x over SP1 and R0VM on L40S GPUs, on CPU (the page carries methodology corrections) | hash-based; no recursion shipped | Apache-2.0 or MIT | **the company shut down 12 Nov 2025**; the only zkVM on it (PetraVM) is archived |
| 2026 entrants | Cysic Venus (a ZisK fork with cudaGraph tuning and an FPGA backend, Apache or MIT, "do not use in production"); Zilkworm (Erigon's C++ guest on Airbender, 2 x 5090 9.3 s); zkDTVM (evmone guest, 4 x 5090 4.7 s, no public docs); Delphinus zkWasm (Halo2 on BN254, a 4090 minimum plus 58 GB of host RAM); Miden (Goldilocks STARK, client-side, Metal via `miden-gpu`, mainnet alpha planned); Boojum (2023 claim of proving on a 16 GB card, superseded by Airbender) | none states a floor under 24 GB on a GPU | | | | |
The reading of the table. No shipped zkVM documents a GPU floor under 24 GB except RISC Zero, whose memory is a
function of a runtime knob (`segment_limit_po2`) and is published per card size by Boundless. SP1's 24 GB is a line
of code, not a property of the proof system: Airbender, OpenVM and Pico all pad to the card they tune on, and all
three say 24 or 32 GB because their market is Ethereum blocks on 5090 clusters. The real-time race has collapsed to 2
to 4 consumer cards per block (ethproofs.org, 5 October 2026), which is why nobody is tuning for a 12 GB card: the
customer buys 5090s. Igneum's customer is the miner who already owns the card, so Igneum has to do the tuning itself.
### 2.2 The sumcheck and GKR family against FRI STARKs, in memory terms
| Family | What it holds at the peak | Blowup | The GPU figure today | Source |
|---|---|---|---|---|
| FRI STARK (RISC Zero, Airbender, ZisK, Pico, Stwo, SP1 3 and 4) | trace, its Reed-Solomon codeword at the blowup, the Merkle trees, the DEEP quotient in the extension field | 4x (RISC Zero, Airbender), 2x (SP1 3 and 4, approximate), 2x (Stwo at rate 1/2) | RISC Zero: 9 to 10 GB per 1 M cycles (Bento) | section 1.5; Boundless pages |
| Sumcheck with a hash-based PCS (SP1 Hypercube, OpenVM SWIRL, Ceno, Ziren) | the trace as multilinears, the extension-field folded copies of the zerocheck and GKR, and the BaseFold or WHIR codeword of the stacked polynomial (still a Reed-Solomon encoding, at 4x in SP1, section 1.1) | 4x of the stacked data in SP1; WHIR's rate is a parameter | SP1: the fixed 13.9 GB plus about 6.5 GB for a 4.7 M-cycle shard (measured) | section 1 |
| Sumcheck with a curve or lattice PCS (Jolt) | the trace and the one-hot columns; **no codeword at all**: Dory commits by MSM and Akita by lattice hashing, so memory is bytes per cycle with no blowup | none | no GPU figure: 200 bytes a cycle on CPU (Akita), so the adopted 4.7 M-cycle shard is about 0.9 GB of prover RAM, approximate (derived) | a16z substack, Sep 2026; the Jolt book, streaming page |
| GKR (Expander, Ceno) | the circuit witness layer by layer; no codeword for the inner layers | none inside; a PCS for the inputs | Expander: 16 MB per Keccak, approximate | Polyhedra blog (returned 530 tonight) |
| Linear-code PCS (Ligero, Brakedown, Ligerito, Blaze) | one encoded matrix and one Merkle tree; linear time, no FFT | rate 1/2 to 1/4 | no prover memory benchmarks found; Linea's Vortex is the only production use | eprint 2021/1043, 2025/1187, 2024/1609; `linea-monorepo/prover/protocol/compiler/vortex` |
| Binius (binary field) | words of GF(2^64) and a BaseFold FRI | 2x to 4x | none; CPU only; company closed | irreducible.com posts |
The memory law in one line: a FRI or BaseFold prover holds `blowup x trace` plus the trace itself plus extension-field
working copies, so 8 to 12 bytes per cell at the peak; a Jolt-class prover holds the trace and its lookups at about 4
bytes per cell and commits without encoding. The figure that matters for us is not the ratio but the absolute: our
adopted shard is small enough (about 280 M cells, section 1.4) that a FRI-class prover sized to it fits a 12 GB card,
and a Jolt-class one fits a phone. The reason SP1 does not fit today is section 1.3, not the proof system.
### 2.3 Folding schemes
| Scheme | Prover memory per step | The verifier at the end | Field | GPU | Fit for Igneum |
|---|---|---|---|---|---|
| Nova, SuperNova, HyperNova, ProtoStar, Mova; Sonobe as the library | one step's witness plus the running instance: tiny by construction (eprint 2021/370) | an IVC proof of O(F) group elements, compressed by a SNARK: Sonobe's decider is Groth16 over BN254 with KZG, about 11.9 M constraints for a 500 k-constraint step (sonobe.pse.dev, decider page); MicroNova about 2.2 M gas (eprint 2024/2099) | curve cycles (Pasta, BN254 and Grumpkin) | partial: sppark MSM on Pasta, a GPL-3 `cuda-nova` for BN254; Sonobe lists GPU as a plan | **no**: a RISC-V step over a 256-bit curve cycle costs two MSMs per step, the opposite of the hash-based consumer-card design decision (design 5.6), and the verifier changes to pairings |
| Nexus zkVM 1 and 2 | the prover ran "on as little as 1 GB of RAM" (whitepaper, approximate) | a curve SNARK | curve cycle | none | **abandoned by its own author**: Nexus 3.0 (25 June 2025) moved to a Circle STARK, "proofs are smaller, faster to generate" (StarkWare blog) |
| LatticeFold, LatticeFold+, Neo, SuperNeo; Nightstream as the zkVM | one step plus an accumulator, lattice commitments over 64-bit fields | a Spartan-class decider | Goldilocks named; BabyBear and KoalaBear not | none; Nethermind's LatticeFold is a "proof-of-concept prototype" whose benches take 48 h; Nightstream is "research software, not production-ready" with its RV32IM prototype removed | **not before 2027 at the earliest**; the first candidate that folds small-field STARK steps |
| Arc, WARP (hash-based accumulation of Reed-Solomon proximity claims) | small: Merkle openings per step (eprint 2024/1731, 2025/753) | a FRI-style accumulator check | any STARK field | none; no public implementation found | the right primitive on paper for folding RISC-V STARK shards with a hash-based verifier; nothing to adopt |
| Mangrove, Nebula | 390 MB peak at 2^24 gates (Mangrove, eprint 2024/416); pay-per-use steps (Nebula) | curve SNARK | curve cycle | none | research |
What folding would mean for a shard: the shard prover would hold one transaction's step at a time and the memory
floor would vanish; the price is a curve-based decider at the end of every shard (seconds on a CPU, a different
verifier in the node, pairings on the light-client path), and no production code over our field. Today's small-field
zkVMs get their bounded memory from segmenting and recursion (2.4), not from folding. Folding is a watch item, not a
route.
### 2.4 Continuations and segment proving at small sizes
| Prover | The segment knob | What a 2^18 or 2^19 segment costs | Can our shard be cut that way inside the guest? |
|---|---|---|---|
| RISC Zero | `segment_limit_po2`, runtime, 13 to 24 (`env.rs:181-186`); the recursion lifts every segment at 2^18 rows and joins them in a tree | po2 19 is the documented fit for an 8 GB card and po2 20 for a 16 GB card (Boundless); the scheduler's measured tokens put po2 18 at about a third of po2 21 and a lift or join at an eighth (`factory.rs`); the 4.7 M-cycle shard at po2 19 is 9 segments, 9 lifts and 8 joins | yes, with no guest change: the zkVM cuts at the limit on its own; the shard statement is unchanged and one succinct receipt comes out |
| SP1 | `HEIGHT_THRESHOLD` (honoured) and `ELEMENT_THRESHOLD` (overwritten by the server, section 1.2); `SHARD_SIZE` up to 2^24 cycles | the live buffers shrink (22.9 GB against 28.3 GB on the prototype shard at `HEIGHT_THRESHOLD 2^20`, bench-log) and the fixed 13.9 GB does not; the compose tree folds 4 proofs at a time | yes, the same way; but the floor is the server's, so the cut buys nothing until the server is re-sized (route A) |
| Our own planner | `S_p` in pgas, a consensus parameter changed by the fee-switch pattern (`docs/plans/fee-switch-devnet.md`); the cut is at transaction boundaries (`core/src/plan.rs`) | halving `S_p` halves the live trace and doubles the shard count; the aggregator verifies one deferred proof per shard (1.66 M cycles for 4 shards, bench-log 4 October) and the chained aggregation is one per block whatever the count (9.7 s on a mining 5090) | yes, already implemented; a transaction above `S_p` stays one shard and the zkVM's own continuations cover it (spec 7.6 item 1) |
| Jolt | none: monolithic; streaming planned | n/a | no |
### 2.5 Distributed proving across several small cards
| System | How one execution is split | Per-card memory | Several cards on one host | What it means for four 12 GB cards |
|---|---|---|---|---|
| SP1 cluster (`sp1-cluster`, BSL 1.1) | by core shard: `ProveShard`, `RecursionReduce`, `RecursionDeferred` and `ShrinkWrap` tasks go to GPU workers, `CoreExecute` and the Groth16 or Plonk wrap to CPU workers (`crates/prover-types/src/lib.rs:31-41`); artifacts through Redis and S3 | "only certain GPUs with >= 24GB RAM are supported" (`infra/charts/sp1-cluster/values-example.yaml`); one task holds one whole card | yes: one GPU node process per card (`gpu{0..7}` services in the docker-compose deployment page); the local server itself supports device 0 only (`task.rs:160`, "only device 0 is supported at the moment"), one server per `CUDA_VISIBLE_DEVICES` | the split is by core shard, and the adopted shard is ONE core shard (section 1.3), so there is nothing to split across cards; the floor per card is unchanged. The cluster is a throughput tool, and its code is BSL |
| RISC Zero Bento (Boundless) | by segment onto Redis; `gpu_prove_agent` spawns one prove agent per card with `CUDA_VISIBLE_DEVICES`; the same `SEGMENT_SIZE` for every card, "the lowest common denominator"; joins form a tree (docs.boundless.network, performance-optimization and bento pages; `compose.yml:62-113`) | by `SEGMENT_SIZE` (2.1): 8 GB po2 19, 16 GB po2 20 | yes, documented: one 16 GB card 264 kHz, two 431 kHz (sub-linear, "bound by bus bandwidth, memory") | **the one documented configuration**: four 12 GB cards at po2 19 or 20 take segments off one queue and the joins fold them; the per-card floor is the segment, and the cost of small segments is the lift and join count (9 lifts and 8 joins for the adopted shard at po2 19) |
| RISC Zero `RISC0_PROVER=actor` | one process, several cards, a token budget per card from measured memory per po2 (`r0vm/src/actors/factory.rs`) | per po2 | yes, experimental since 3.0.1 | the same model without Bento's services |
| Pico Prism 2.0 | a global task queue across two machines, 16 x 5090, 100 Gbps between them (Brevis blog, May 2026) | not published | yes | no figure |
| OpenVM | metered execution on the CPU, segments to GPUs, an aggregation tree, "clusters with hundreds of GPUs" (docs, distributed-proving page) | 24 GB | yes | no 12 GB path |
| ZisK | coordinator and stateless workers; "splits the trace into pieces, proves each in parallel on separate machines, and aggregates"; the first worker aggregates a binary tree (docs, distributed execution page) | not documented | yes, `--gpu` per worker | no figure |
| Ceno | shards round-robin by `shard_id % device_count`; a shard never split across cards; one CUDA context per device (PR #1403) | not stated | yes | the same model |
| Column-split of one trace across cards (FRIttata eprint 2025/1285, HyperFond 2025/1349, deVirgo arXiv 2210.00264, Pianist 2023/1271, Cirrus 2024/1873, SumFold 2025/1653) | the sumcheck or FRI itself is distributed, each worker holding a slice of the columns or rows and exchanging small messages | a slice | research code or CPU clusters only | nothing shipped; the one route that would let four 12 GB cards hold what one 32 GB card holds for a SINGLE core shard, and nobody has it in a zkVM |
The reading. Every shipping system splits by rows (segments, shards, chunks), proves each on one card, and folds
with recursion. So "four 12 GB cards do what one 32 GB card does" is true for throughput (four shards in flight, or
four segments of one shard, then a join tree) and false for a single unit that exceeds one card: that unit must be cut
smaller, by the zkVM's segment knob (RISC Zero) or by our planner (`S_p`). For Igneum the units are already small and
independent (a shard, assigned by sortition), so the rig's natural mode is one prover process per card, each taking
its own shard. The distributed route therefore costs nothing in protocol and lands as an app change (section 3, route C).
### 2.6 Proof systems that run on AMD or Apple
| Target | What exists | Status | Source |
|---|---|---|---|
| Apple, RISC Zero Metal | the full STARK prover (rv32im, keccak, recursion) on Metal, automatic on Apple silicon; the Groth16 wrap x86 only | shipped, maintained (PR #3761's June 2026 matrix lists "metal (Mac M-series): build, run"); the only speed published is a 2023 M2 datasheet (14 to 93 kHz, approximate); nothing for M3, M4 or M5 | `risc0/sys/kernels/zkp/metal/`, dev.risczero.com local-proving page |
| Apple, ICICLE Metal (Ingonyama) | MSM, NTT, sumcheck on Metal since v3.6.0 (Mar 2025); "missing API implementations for Poseidon and Poseidon2 hashes, Merkle tree" at that release; v4.0.0 of 11 Jul 2025 is the latest | a library, closed-source backends under a free research licence (dev.ingonyama.com, install_gpu_backend page); no STARK prover built on it for Metal | ICICLE releases, the Metal blog |
| Apple, Jolt Metal | PR #1938 merged 30 Sep 2026 (the runtime and field kernels); PR #1733, the prover itself, a draft: M5 Max 2^25 cycles in 19.8 s, 3.2x over its CPU | the fastest Apple number anyone has published, in a draft | github.com/a16z/jolt pulls 1733 and 1938 |
| Apple, Stwo | CPU SIMD with NEON; ICICLE-Stwo promises Metal | CPU path shipped; no RISC-V guest of its own | stwo README, Ingonyama blog |
| Apple, Miden | `miden-gpu` on Metal | Cairo-class VM, not RISC-V | hackmd (bobbinth) |
| AMD, sppark | "A limited support for AMD's RDNA and CDNA GPUs" (README); SP1's tree carries no HIP build | a library | github.com/supranational/sppark |
| AMD, SP1 PR #2668 | an external port to RDNA3 and RDNA4 with "a caching memory allocator to work around hipMallocAsync leak bug" | **closed unmerged 20 Mar 2026** | github.com/succinctlabs/sp1/pull/2668 |
| AMD, OpenVM stark-backend HIP fork | `cuda2hip.hpp` so the same `.cu` builds under nvcc and hipcc, native `mont32_t.hip`, tested on gfx1100, targets MI300X and 7900 XTX | merged 15 Sep 2026 in a fork (Okm165/stark-backend PR #2), not upstream | the PR |
| AMD, Goldilocks NTT and STARK on ROCm | 19.19 ms NTT at 2^27 on an RX 7900 XTX; a Goldilocks STARK backend on HIP | research posts | ethresear.ch, qingming-g64-ntt and stark-g64 |
| Vulkan and WebGPU | ICICLE's Vulkan build (Jan 2025) with no installable backend; zkSecurity's WebGPU Stwo (5x on constraint evaluation, 2x end to end, no 64-bit integers in WGSL); ZPrize WebGPU MSM | prototypes; nothing proves a RISC-V shard | the pages named |
Said plainly, as `docs/analysis/amd-proving.md` said it: on 5 October 2026 no zkVM proves on an AMD GPU, and the only
Apple prover that ships is RISC Zero's. The AMD work that exists is two ports of CUDA STARK kernels through a HIP shim,
one closed, one in a fork; both are days of agent work to revive against a given tree, and PC 1's RX 9070 XT (gfx1201)
is the card to measure on.
## 3. For each route: the change to our guest, the aggregator and the node's verifier; the cost; the risk; 12 GB under 60 s
What the node verifies today: SP1 compressed proofs through `igneum-prove-host --mode verify` and `verify-segment`
(`vendor/igneum-node-pv1/igneum/exec/src/proving.rs:41-46, 878-926`), the pinned ids read at start and named in the
native statement (`program_ids`, `IGNEUM_PROOF_PROGRAM_IDS`), the record bound in a BLS-signed `ProofRecord` (version
1) or `SegmentRecord` (version 2) with the proof's SHA-256 (spec 7.7 item 1, 7.8 item 3). A different proof system
means a new `ProofSystem` version (design 5.6), a new pinned id, a second verifier command, and the record's version
field telling the node which. The swap procedure of design 5.6 (test vectors, 90% signalling, a 3-month overlap with
both verifiers, a wrap of the last old proof) is the path for any of the rows below that change the family.
The 60-s test. The litepaper's minute, the launch target of 20 to 60 s behind the tip, and the mine-and-prove
measurement that a shared card proves 3 to 4x slower (bench-log, `chain-pc2-pv1c`). No 12 GB card has run any
prover in this repository; the 12 GB times below are approximate, scaled from the 5090 by memory bandwidth (an RTX
3060 at 360 GB/s and an RTX 4070 at 504 GB/s against the 5090's 1,792 GB/s, NVIDIA's published figures, approximate),
which is the term a STARK prover is bound by. They are the numbers the first 3060-class run replaces.
| Route | Guest | Aggregator | Node verifier and record | Cost (agent time) | Risk | Reaches 12 GB with a real shard under 60 s? |
|---|---|---|---|---|---|---|
| **A. Re-size SP1's GPU server** (the prover-floor agent's patch, running tonight): remove the 20 GB panic (`builder.rs:37`), size `max_trace_size` to the shard (honour `ELEMENT_THRESHOLD`, or set the core allocation from the shard's measured cells), one core worker and a buffer of 1, a release threshold so the pool returns memory between stages, `drop_ldes` on; build with `CUDA_ARCHS` for Ampere, Ada and Blackwell | none: the same ELF, the same pinned id (the verifying key hashes the program and its preprocessed tables, not the server's buffer sizes; `HEIGHT_THRESHOLD` only shortens tables below the verifier's 2^22 maximum) | none: the compressed proof format and the aggregator guest are unchanged | none: the same `--mode verify`; the record format unchanged | hours to one day: a fork of `sp1-gpu/crates/prover_components` and `jagged_tracegen` (Apache or MIT), the 11-min cross-build, a per-card profile in `provedefault.rs`, a CI check that the fork's constants match the pinned verifier's | low on the protocol, medium on the build: the fixed recursion stage may hold the floor near 8 to 9 GB (section 1.3, approximate) and the first measurement says whether 11 GB is reached; a fork of `sp1-gpu` to carry forward on every SP1 release; the server rejects nothing it cannot hold, so an out-of-memory shard must fail cleanly and be left (the pool's rule today) | **memory: likely for the adopted shard** (10 to 11 GB on paper, section 1.4), **not** for the prototype shard (28 GB of live trace). **Time: prove-only yes** (4.3 s on the 5090 scales to about 15 to 22 s on a 3060 and 10 to 15 s on a 4070, approximate); **mine-and-prove on a 12 GB card: no at `S_p`** (3 to 4x on a shared card puts a 3060 at 45 to 90 s, approximate, and the miner's 1.7 GB on top of 11 GB does not fit), yes at `S_p/2` on a 4070 if the floor lands under 9 GB (approximate). The measurement decides; this is the route the gate waits on |
| **B. Halve `S_p`** (30,000 to 15,000 pgas, the fee-switch pattern): more and smaller shards | none | none: one deferred proof per shard, so 2x the shards per block; the chained aggregation stays one per block (9.7 s mining, 2.5 s alone) | none | hours: a fee-table change and a rollout plan like `fee-switch-devnet.md` | low: more records per block (the coinbase carries at most 8 shard records, spec 7.7 item 2, so `B_p / S_p` must stay at 8 or under); the assignment window and sortition unchanged | **alone, no**: the floor is the server's (13.9 GB at 0 cycles). **With A, it is the dial** that moves a 12 GB card from prove-only to mine-and-prove, and a 16 GB card to a comfortable fit |
| **C. One prover process per card on a rig** (the distributed route): the app runs one `sp1-gpu-server` per NVIDIA card (`CUDA_VISIBLE_DEVICES`, the per-device socket of `sp1-cuda/src/client.rs:211`, `.cuda().with_device_id(n)`), one host process per card, each taking its own assigned shard; the rig installer already picks cards (`igneum-rig-lib.sh`, `prover_decision`) | none | none: shards are independent units by design (spec 7.2); the aggregator runs on the biggest card | none | one day: the app's prover loop per card (`prover.rs` runs one loop today), the Settings and tile per card, the rig installer's prover unit per card, the socket cleanup per device (the root-socket rule of 5 October) | low; the throughput is per card, the host RAM 6 GB of pinned buffers per server (section 1.2), so a 4-card rig needs 32 GB of RAM or route A's smaller buffers | **it does not move the floor**: each card still needs A. It is the route that makes four 12 GB cards worth four shards a cycle, and it ships with A, not instead of it. Splitting ONE shard across cards is not a route: the adopted shard is one core shard (2.5), and column-split provers are research |
| **D. RISC Zero as proof system version 2** (CUDA and Metal; segments at po2 19 or 20) | a second guest: `core/` is plain Rust and ports as is; the precompile patches differ (SP1's `sha3` and `k256` patches against RISC Zero's `sha2`, `k256` and keccak circuit); the shard statement bytes unchanged; a second pinned ELF and image id in `elf/manifest.json` | a RISC Zero aggregator guest using composition (`env::verify` of the shard receipts, dev.risczero.com composition page); the chain rule (N verifies N-1) inside the family; **a block's shards must be one family**, and a chain cannot cross families inside the proof: a family switch lands at a segment boundary as a fresh chain (spec 7.8 item 6 already allows one after an unproven segment; the rule gains "or at a proof-system version change") | a second verifier mode (`--mode verify-r0`, the `risc0-zkvm` verifier, pure Rust, about 100 ms, 222 KB receipts); the record's `version` selects the family; the native statement names the family's pinned id; both verifiers in the node through the overlap of design 5.6 | 3 to 4 days: guest port and pinning 1, aggregator and chain rule 1, node verifier and record version 1, app profile and host modes 0.5, test vectors and the fast-time harness 0.5; plus the measurement day on PC 2 | medium: two proof systems in consensus for the overlap; RISC Zero's Groth16 wrap is x86 only (the light-client path of ledger P3 stays on SP1 or waits); a 222 KB receipt per shard against 1.27 MB today is a gain; the recursion tree per shard (9 lifts and 8 joins at po2 19) is extra time on small cards; `main` is at 5.0.0 with no release body, so the pin is 3.0.6 | **memory: yes by documentation** (po2 19 for an 8 GB card, po2 20 for 16 GB; 9 to 10 GB per 1 M cycles), the first documented sub-12 GB prover. **Time: approximate**: a 4090 does 808 kHz at po2 21, so the adopted shard is about 6 s on a 4090-class card and about 20 to 30 s on a 3060-class one at po2 19, prove-only; beside the miner over 60 s on a 3060, near it on a 4070. The prover-floor agent's PC 2 run is the first real number |
| **E. Airbender, OpenVM, ZisK, Pico, Ziren** as version 2 | a new guest each (RISC-V, except Ziren's MIPS); OpenVM's and ZisK's toolchains are the most complete | each has its own recursion; OpenVM's aggregation and Halo2 wrap are the most documented | a new verifier each (STARK under 300 KB for OpenVM; PLONK or FFLONK for ZisK and Airbender) | 4 to 6 days each | the same two-family cost as D with no memory gain: 21 GiB (Airbender), 24 GB (OpenVM, Ziren), undocumented (ZisK, Pico); Pico's and Ziren's GPU code is BUSL or closed | **no**: none documents a floor under 21 GiB; the race is tuned for 5090 clusters |
| **F. Jolt (Lattice Jolt) as version 2**: a sumcheck prover with no codeword; CPU and Metal | a new guest (RV64IMAC, Jolt's toolchain; no keccak precompile today, approximate, so the trie hashing costs more cycles than in SP1) | **none exists**: no recursion or continuation shipped, so the aggregator would verify N shard proofs natively and the chain rule would live in the native statement until Jolt's recursion lands | a Dory verifier (BN254 pairings, about 50 KB, sub-second, approximate) or an Akita verifier (lattice, 65 to 80 KB); no on-chain verifier shipped | 5 to 8 days for the guest, the verifier and the record; the aggregator question has no answer in the code | high: alpha software, no audit, no production user, no recursion; the proof system of the miner's CPU, not of its card | **memory: yes by a wide margin** (about 0.9 GB for the adopted shard at 200 bytes a cycle, approximate). **Time on a CPU: about 2 to 3 s** for 4.7 M cycles at over 2 M cycles a second (a16z, Sep 2026, laptop CPU; approximate for our guest), **on Metal under 1 s** (PR #1733's 2^25 in 19.8 s on an M5 Max, approximate). The numbers are the best in this document and the software is not shippable |
| **G. Folding** (Nova family, lattice folding) | a step circuit per transaction or per opcode group | a decider per shard | pairing or lattice verifier | weeks of research, no code over our field | the family that Nexus left | **no** today; the watch item for 2027 |
| **H. AMD through a HIP port of SP1's kernels** (PR #2668 revived against 6.8.1, or the `cuda2hip` shim of the OpenVM fork) | none | none | none: the same SP1 proofs | 3 to 5 days plus PC 1's RX 9070 XT to measure; the `hipMallocAsync` leak needs the caching allocator the PR carried | medium: a kernel port with no upstream; the sppark NTT has a limited HIP path and cuPQC none | memory as route A (the same buffers); **time unmeasured on any AMD card**; the one route that gives AMD miners the 20% pool share |
| **I. Apple through RISC Zero Metal** (route D's Metal half) | as D | as D | as D | inside D's 3 to 4 days | the 2023 M2 figure (14 kHz, approximate) says 5 minutes for the adopted shard; an M5 Max is not measured by anyone | **memory: yes** (unified memory, 64 GB on the M5 Max). **Time: unknown**; the Mac measure lock run is the number |
## 4. The ranked recommendation
| Rank | Route | Why | Gate |
|---|---|---|---|
| **1. Soonest to 12 GB with the least change: A, with B as the dial and C for rigs** | re-size SP1's GPU server; keep the guest, the aggregator, the verifier and the pinned ids exactly as they are; set `S_p` from the first 12 GB measurement; one prover per card on rigs | nothing in consensus moves; the work is a fork of two Apache crates and an app profile; it is already running tonight; every other route costs days and adds a second verifier | the prover-floor agent's rows: the adopted shard under 11 GB alone and the time on the first 3060-class or 4070-class card, prove-only and beside the miner. If under 11 GB and under 60 s prove-only: ship 0.3.12 with the 12 GB tier as prove-only and `S_p/2` measured for mine-and-prove. If not under 11 GB: route D |
| **2. The fallback if A misses 11 GB, and the Apple route either way: D, RISC Zero as version 2** | the only shipped prover with a documented sub-12 GB configuration and a shipped Metal path; Apache or MIT including the kernels; 222 KB receipts | the swappable interface was built for this (design 5.6) and the node already names the pinned id in the statement, so a second family is a version, not a redesign; the cost is 3 to 4 days plus the overlap | PC 2's po2 19 and 20 rows (memory, time per segment, lift and join) tonight; the Mac's Metal row |
| **3. Best in five years: the sumcheck family without a codeword (Jolt-class), or the sumcheck-plus-WHIR family SP1 and OpenVM already converge on** | Jolt proves the adopted shard in seconds on a laptop CPU at under 1 GB of memory, which is the only route that gives AMD-only, Apple and 8 GB machines the proving share with their existing hardware; its verifier is small (50 to 80 KB); its licence is MIT or Apache. It is alpha with no recursion, so not before it ships a stable release with continuations and an audit. SP1 Hypercube and OpenVM SWIRL are the same mathematics with a hash-based PCS and a GPU today, which is why staying on SP1 now loses nothing in that direction | do not adopt now; re-read Jolt and the Arc or WARP accumulation line at every 6-month era draw (design 5.6's swap procedure needs 90% signalling and a 3-month overlap, so the lead time is the schedule) | a stable Jolt tag with recursion, an audit, and a CUDA or merged Metal prover |
| **The interface question** | yes: `ProofSystem` is versioned (`VERSION`, `program_id`, `verify_segment`), the record carries `version`, the node reads pinned ids at start and names them in the native statement, and the overlap procedure keeps both verifiers in the node for 3 months with `B_p` from the stricter table. What is missing for two families at once is small and named: the record version selecting the verifier command, the fresh-chain rule at a version change, and the shard plan carrying the family per block so a block's shards are homogeneous (the aggregator folds one family). Those three items are in route D's day of node work | so the answer to "ship one now and move to the other later" is yes, and route A ships nothing that has to be undone | |
The honest statement of what this ranking does not know: no 12 GB card has run any prover here. Route A's time
figures are bandwidth scaling, labelled approximate; route D's are a 4090 figure scaled the same way. The first 3060
or 4070 in this repository replaces both columns, and the plan is to borrow or buy one this week (a 4070 is the
common 12 GB card of 2026; a 3060 the common older one; both are the gate's named class, design R2).
## 5. The tier consequences, and the public line while the change is made
Every number carries its consequences (CLAUDE.md, 5 October 2026). The table says what each tier has today on SP1
6.8.1, what route 1 (A plus B plus C) gives it if the gate is met, what route 2 (D) adds, and what only route 3 would
give. "Today" is measured; the rest is the routes' expected outcome, labelled, until the measurement.
| Tier | Today (measured, bench-log 5 October) | Route 1: re-sized SP1 server, `S_p` as the dial, one server per card | Route 2: RISC Zero version 2 | Only route 3 (sumcheck without a codeword) |
|---|---|---|---|---|
| Home miner, one 8 GB NVIDIA card | mines; proves nothing (the server panics under 20 GB) | proves nothing at `S_p` (the floor's fixed terms, 8 to 9 GB approximate, leave no room); perhaps empty shards | prove-only at po2 19 (Boundless' 8 GB tier), the miner paused per shard; time approximate 30 to 60 s | mines and proves on its CPU |
| Home miner, one 12 GB card (3060, 4070) | mines; proves nothing; the litepaper's gate card | **prove-only at `S_p`** if the floor lands under 11 GB (expected, section 1.4): about 15 to 22 s a shard, approximate; **mine-and-prove at `S_p/2`** on a 4070 if the floor is under 9 GB, approximate; on a 3060 the shared card misses 60 s, approximate, so its default is prove-only with the miner paused per shard (the 16 GB rule of `provedefault.rs` today, moved down a tier) | prove-only at po2 20 (16 GB tier) or po2 19; mine-and-prove not inside 60 s on a 3060, approximate | mines and proves, CPU |
| Home miner, one 16 GB card (5080, 4080, 4060 Ti 16 GB) | an empty shard alone (13.9 GB); nothing beside the miner | **mine-and-prove at `S_p`** (11 GB plus the miner's 1.7 GB), about 7 to 12 s a shard alone and 20 to 40 s beside the miner, approximate | mine-and-prove at po2 20 | the same |
| Home miner, one 24 GB card (4090, 3090) | the adopted shard alone (20.4 GB) and beside the miner (22.2 GB, approximate for the card); the prototype shard never | mine-and-prove at `S_p` with 10 GB to spare; the prototype shard (28 GB live) only if the devnet's fee switch has passed, which it has from DAA 210,000 | the same with Metal irrelevant | the same |
| Home miner, one 32 GB card (5090) | everything, measured | everything, with more shards in flight if the pool releases between stages | the same | the same |
| Rig, several NVIDIA cards | one prover on the biggest card (`prover_decision`) | **one server per card**, each its own shard; the aggregator on the biggest card; host RAM 6 GB pinned per server today, under 2 GB with route A's buffers | the same model (Bento's) | the same |
| Pool user | through the pool; who proves is open (spec 09) | unchanged | unchanged | unchanged |
| AMD-only (RX 9070 XT, 7900 XTX) | mines; proves nothing on the card; the CPU path 282 s a shard at 30 GB | unchanged until route H (a HIP port, 3 to 5 days, measured on PC 1's 9070 XT) | unchanged: RISC Zero is CUDA and Metal only | mines and proves on its CPU |
| Apple silicon (M-series) | mines (26.7 MH/s on the M5 Max); the SP1 CPU prover 41 to 55 s for an empty shard, 272 s for a small one | unchanged | **proves on the GPU through Metal** (64 GB unified memory on an M5 Max holds any segment); the time is the measurement | proves in seconds on Metal (Jolt's draft PR figure, approximate) |
| Windows under 32 GB of RAM | off (the WSL2 prover held 7.9 GB) | the pinned buffers fall with `max_trace_size`, so a 16 GB PC likely qualifies, approximate; measure | RISC Zero's CUDA path also runs in WSL2 | |
The deadlines these fit (spec 7.2 item 3, the litepaper, `proving-v1.md`): the 10-s exclusive window is the 5090's
alone; a 12 GB card at 15 to 22 s proves its assigned shards in the open phase and is paid when no faster card took
them, which on a chain with few 5090s is most of the time; the minute of the litepaper holds for prove-only 12 GB
cards and for mine-and-prove 16 GB cards; the 600-s unproven deadline holds for every tier above the CPU path.
### The public line while the change is made
The litepaper's sentence today ("Target: shard size will be set so a 12 GB card proves one shard in about 20
seconds") is a target and says so (fud-ledger P1, overclaim 27). What this document adds, for `site/litepaper.html`,
`site/miner.html` and the app's Proving tile, in the copy law:
> Proving runs on NVIDIA cards with 24 GB or more today. A build for 12 GB and 16 GB cards is being measured: the
> memory is the prover's buffers, not the shard, and the fix is a smaller build of the same prover. AMD and Apple
> cards mine. A second prover with an Apple path exists and is the fallback.
And the rule for the next status line, whichever way the measurement goes: the number, the card it was taken on, and
the tier it moves, in one sentence, the day it is taken.
### What this document does about it
| Consequence | Action | Owner |
|---|---|---|
| The gate card has never run a prover here | get a 4070 or 3060 into the measurement loop this week; until then every 12 GB figure stays approximate | coordinator; the project lead for the card |
| Route A's gate | the prover-floor agent's rows (asked for by message tonight); if under 11 GB, `provedefault.rs` gains the 12 GB prove-only and 16 GB mine-and-prove tiers and the rig installer one server per card | prover-floor agent, then the proving engineer |
| Route D's measurement | RISC Zero 3.0.6 at po2 19 and 20 on PC 2 (CUDA) and on this Mac (Metal), the same shard statement run natively: memory, time per segment, lift and join, receipt size | prover-floor agent (PC 2); a Mac measure job for Metal |
| The two-family node items (record version selects the verifier, fresh chain at a version change, one family per block) | spec 7.8 gains the three rules when route D starts; nothing changes before | execution engineer |
| AMD | route H is a 3-to-5-day job with a measurement on PC 1's 9070 XT; opened as a plan when route A's result is in | execution engineer |
| The public line | the paragraph above to the site and the tile with the next site pass | site-pages owner |
## Sources
Our own: `docs/bench-log.md` entries "proving v1: segment records, the chain rule, the unproven rule" (5 October 2026),
"the SP1 CPU prover on PC 1" (5 October), "shard proving on the RTX 5090" (4 October); `docs/plans/proving-v0.md`,
`proving-v1.md`; `docs/analysis/amd-proving.md`; `docs/spec/07-execution.md` 7.2, 7.6, 7.7, 7.8; `docs/design/execution-layer.md`
5.1 to 5.7; `proving/igneum-prove` (`host/src/proof_system.rs`, `program/src/main.rs`, `aggregator/src/main.rs`,
`elf/manifest.json`); `vendor/igneum-node-pv1/igneum/exec/src/proving.rs`; `app/igneum-app/src/provedefault.rs`, `prover.rs`.
SP1 6.8.1, read from `~/.cargo/registry/src/index.crates.io-*/` and the vendored tree `vendor/sp1-6.8.1` (commit
c84ada1e, 24 Sep 2026) on the `prover-floor` worktree: `sp1-core-executor-6.8.1/src/opts.rs`, `src/utils.rs`,
`src/artifacts/rv64im_costs.json`; `sp1-prover-6.8.1/src/components.rs`, `src/worker/config.rs`, `src/shapes.rs`;
`sp1-primitives-6.8.1/src/fri_params.rs`; `sp1-verifier-6.8.1/src/compressed/config.rs`; `sp1-hypercube-6.8.1/src/verifier/config.rs`;
`sp1-cuda-6.8.1/src/server.rs`, `src/client.rs`; `sp1-gpu/README.md`, `sp1-gpu/crates/prover_components/src/builder.rs`,
`src/components.rs`, `sp1-gpu/crates/jagged_tracegen/src/lib.rs`, `sp1-gpu/crates/shard_prover/src/prover.rs`,
`sp1-gpu/crates/cuda/src/task.rs`, `src/device.rs`, `sp1-gpu/crates/sys/lib/runtime/mem_pool.cu`, `sp1-gpu/crates/zerocheck/src/primitives.rs`.
Web: docs.succinct.xyz (hardware-acceleration, hardware-requirements, proof-types, security-model, provers introduction,
cluster architecture, docker-compose deployment); blog.succinct.xyz (sp1-hypercube, real-time-proving-16-gpus,
sp1-hypercube-is-now-live-on-mainnet); github.com/succinctlabs/sp1 releases v6.0.0 to v6.8.1, issues #2674, #2930,
#2950, #2969, pulls #2631, #2668, #2723, #2917, #2974; github.com/succinctlabs/sp1-cluster (README, LICENSE,
`infra/charts/sp1-cluster/values-example.yaml`, `crates/worker/src/config.rs`); eprint 2025/917 (jagged polynomial commitments).
RISC Zero: `~/.cargo/registry` crates `risc0-zkp-3.0.4/src/lib.rs`, `risc0-zkvm-3.0.4/src/receipt.rs`, `src/host/recursion/prove/mod.rs`,
`risc0-circuit-rv32im-4.0.4/src/execute/mod.rs`, `src/zirgen/defs.rs.inc`, `src/prove/hal/cuda.rs`; github.com/risc0/risc0
`risc0/zkvm/src/host/client/env.rs`, `risc0/zkvm/Cargo.toml`, `risc0/zkvm/build.rs`, `risc0/sys/kernels/zkp/{cuda,metal}/`,
`risc0/r0vm/src/actors/factory.rs`, `risc0/circuit/recursion/src/lib.rs`, releases v2.0.0, v3.0.1, v3.0.6, pull #3761;
dev.risczero.com (local-proving, composition); docs.boundless.network (bento, performance-optimization, quick-start);
github.com/boundless-xyz/boundless (`compose.yml`, `bento/README.md`, `bento/LICENSE-BSL`); github.com/ekrembal/gsr-stark-verifier pull 5; l2beat.com/zk-catalog/risc0.
Others: zksync.io/airbender, docs.zksync.io airbender and proving pages, github.com/matter-labs/zksync-airbender (README,
`docs/gpu.md`, pull #448), veridise.com (the Airbender audit); 0xpolygonhermez.github.io/zisk (introduction, limits,
distributed execution, installation), github.com/0xPolygonHermez/zisk (README, pull #1238, `zisk-contracts`);
blog.openvm.dev (2.0, 2.0-production, 2.1, openvm-gpu, v1), docs.openvm.dev (security-model, distributed-proving, sdk);
pico-docs.brevis.network, github.com/brevis-network/pico and pico-gpu (README, LICENSE), blog.brevis.network (Prism 1.0,
2.0, 2.1); docs.zkm.io (prover, performance), github.com/ProjectZKM/Ziren, zkm.io (the independent evaluation of v1.1.4),
eprint 2026/2330; github.com/starkware-libs/stwo and stwo-cairo (README), ingonyama.com (ICICLE-Stwo, the Starknet
partnership, ICICLE Metal v3.6), dev.ingonyama.com (install_gpu_backend), blog.zksecurity.xyz/posts/webgpu, starkware.co
(S-two 2.0.0, Nexus on S-two), theblock.co (S-two on Starknet); github.com/a16z/jolt (README, book: intro, dory, akita,
streaming, recursion, blindfold; pulls #1733, #1938; tags), a16zcrypto.substack.com ("How to prove software ran
correctly", Sep 2026), a16zcrypto.com (jolt-6x-speedup, 64-bit-proving-jolt, zkvm-jolt-zero-knowledge, faqs-on-jolts-initial-implementation),
eprint 2025/611; github.com/scroll-tech/ceno (README, Cargo.toml, pull #1403), ceno-gpu-mock, scroll.io (Ceno post),
osec.io ("zkVMs' unfaithful claims"); github.com/nexus-xyz/nexus-zkvm (README, LICENSE), blog.nexus.xyz (roadmap);
lita.gitbook.io (Valida architecture, benchmarks); github.com/powdr-labs/powdr; irreducible.com (announcing-binius64,
reinventing-irreducible, irreducible-shutting-down), github.com/binius-zk/binius64, eprint 2026/1656; eprint 2021/1043,
2022/1010, 2024/1609, 2025/1187, 2024/1586, 2024/185 (linear-code commitments, WHIR, Vortex), github.com/Consensys/linea-monorepo;
PolyhedraZK/Expander and blog.polyhedra.network (returned 530 tonight); eprint 2021/370, 2024/2099, 2024/1220, 2024/416,
2024/1605, 2025/247, 2025/294, 2026/242, 2024/1731, 2025/753, 2026/1371 (folding and accumulation), sonobe.pse.dev,
github.com/privacy-scaling-explorations/sonobe, NethermindEth/latticefold, LFDT-Nightstream/Nightstream; eprint 2023/1271,
2024/1208, 2024/1873, 2025/1349, 2025/1653, 2025/1285, 2018/691, arXiv 2210.00264, 2602.16338 (distributed proving);
github.com/supranational/sppark, github.com/Okm165/stark-backend pull 2, ethresear.ch (qingming G64 NTT and STARK on ROCm);
github.com/cysic-labs/venus, erigon.tech (Zilkworm), github.com/DelphinusLab/prover-node-docker, hackmd.io/@bobbinth
(Miden), ethproofs.org/clusters (5 October 2026).

View file

@ -0,0 +1,408 @@
# Layer 3 soundness: the per-warp scratch with read-modify-writes
5 October 2026 (night), cryptographer role, Counter ASIC 2.0 plan step 4 (`docs/plans/counter-asic-2.md`). Branch
`ca2-soundness` on top of `readwidth` b970dda (the scratch as a class parameter, 32 or 128 KiB per warp). Tests:
`igneum-pow/tests/scratch.rs`; Metal runs through `proto-metal/packbench` on the M5 Max; commands and counts in
`docs/bench-log.md` (entry of the same date). Nothing here touches the lottery hash as shipped: variant 5 is behind
`LoadClass::scratch(k, kb)` and is never emitted by generator version 2.
Every figure below is measured (machine, date, command named) or cited; "approximate" marks a figure from memory.
## 0. The five findings
| # | Question | Finding | Status |
|---|---|---|---|
| 1 | Is what is written uniform and beyond a chip's precomputation? | The fill is a bijection of the lane nonce, the rewrite a bijection of the fold value in each word; written words show no bit bias over 3 to 12 million rewrites per class (worst 3.63 sigma of 6). The fill IS precomputable, by design, and at 64 slots 78.5 percent of reads are fill reads. | sound as a function; see 2 for what that means |
| 2 | Does any short cut avoid the writes? | No short cut inside a unit: a slot after d read-modify-writes needs all d fold values (replay test). But the live state is bounded by the read-modify-write count, not by the scratch size, because CPU verification resets the scratch per unit: 64 to 320 bytes per lane at scr2 to scr8, whatever the nominal 32 KiB, 128 KiB or 1 MiB. The named chip (cache mirror plus recompute) keeps that in SRAM at under 5 percent of its mirror and its gain does not move at any share under the 6 GB cap. | NOT sound as an anti-chip layer |
| 3 | Is the verifier's one-warp simulation exact? | Exact when the GPU's lazy per-unit tag is unique over the arena's life and the arena holds no stale tag. The kernels rely on this and neither host guarantees it (no clear at allocation, no clear at the 32-bit wrap of the tag counter, 16.4 minutes on a 5090). With the host contract of section 4.3 the simulation is exact: 14 edge packs twice, 200 fuzz packs, consecutive units on one warp and the wrap inside a launch all match the CPU on Metal (228 of 228); a broken tag and a broken fill are caught (3 of 3). | sound with a host contract; today it is luck |
| 4 | The attack surface of the writes | Out of bounds: impossible by the mask, 42 of 42 emitted kernels pass the static check, which catches six deliberate breaks. Aliasing: none, lane-major arenas disjoint by (warp, lane), two logical units of a wave64 get two arenas. Ordering: one lane, one slot, program order; no cross-lane sharing, no atomics needed. Alignment: 16-byte slots at 16-byte offsets from a 256-byte-aligned base. Wrap: identical to the CPU, tested at the launch level. | sound |
| 5 | What a conformance vector must carry | The class and geometry, the fill and rewrite, the host contract (tags, clearing, groups a multiple of warps), two consecutive units on one warp with a forced slot collision, a unit in the top 256 nonces with the wrap inside the launch, and the fingerprint declared independent of the warp count. The standard three-unit vectors catch a broken tag only through base 1,000,000 and would miss it at a 1 MiB scratch. | defined in section 6 |
Recommendation (section 10): do not adopt layer 3 as the plan states it (read-modify-writes taken from the 16
dataset loads). It replaces latency-bound dataset reads with cache-bound ones for the GPU, costs the named chip
nothing it cannot keep in a few megabytes of SRAM, and leaves that chip's gain at 2.4x at every share. The lever
that moves that chip is the mixer multiplier of the M16 analysis (x2 brings it to 1.2x, x4 to 0.6x, under the
verifier's 10 ms gate). If a scratch is kept for another reason, add the read-modify-writes beside the 128 loads,
never in their place, and ship the host contract and the vector of section 6 with it.
## 1. What the branch implements
| Piece | Where | What |
|---|---|---|
| Class | `igneum-pow/src/generator.rs:170-230` | `LoadClass { scratch: Some(k), scratch_kb }`: `k` of the 16 memory slots are `Op::Scratch`; `scratch_kb` KiB per warp of 16-byte slots, lane-major, `slots = kb x 2` per lane (32 KiB: 64, 128 KiB: 256); `scratch_slot_mask() = slots - 1` |
| Draw | `generator.rs:488-491` | the first `k` of the 16 drawn load slots become scratch ops (a uniform k-subset); the source register follows the fresh-source rule like a load |
| Fill | `igneum-pow/src/verify.rs:30` | `scratch_fill(seed, base, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot x 0x9e3779b1 + (j + 1) x 0x85ebca77)`, j in 0..2 |
| Fold | `verify.rs:18` | `x = dst ^ w0; x = (rotl(x, 11) x 0x9e3779b1) ^ w1; x = (rotl(x, 11) x 0x9e3779b1) ^ w2; dst = x` (the read-width fold over the three data words) |
| Rewrite | `verify.rs:41` | the slot becomes `(x ^ w1, rotl(x, 7) ^ w2, x + w0)` |
| CPU model | `verify.rs:48-100`, `:305-312` | `ScratchModel`: per (lane, slot) a written bit and three words; an unwritten slot reads as its fill; one model per unit, so a unit starts from the fill |
| Acceptance | `igneum-pow/src/accept.rs:202-215, 374` | a scratch site that reads one slot in all 32 lanes rejects the program (lane-constant site); scratch slots carry bit 31 in the address list and are left out of the distinct-address bound, which now covers the dataset loads only |
| GPU statement | `igneum-pow/src/emit.rs:143-157` | `s_ = rN & mask; v_ = 16-byte load of slot s_; m_ = (v_.x == tag) ? ~0 : 0; w = (v_.yzw & m_) \| (fill & ~m_); fold; dst = x_; 16-byte store of (tag, x_ ^ w1_, rotl(x_, 7) ^ w2_, x_ + w0_)` in Metal, CUDA and OpenCL |
| Persistent prologue | `emit.rs:159-175` | `lane = tid & 31; warp_ = tid >> 5; arena = scratch + (warp_ x 32 + lane) x words_per_lane; for (g_ = warp_; g_ < groups; g_ += nwarps_) { gbase = baseNonce + g_ x 32; tag = salt + g_; ... }` |
| Hosts | `proto-metal/packbench.swift:144-164`, `proto-opencl/host.c:1025-1033, 1268` | the arena is allocated and never written by the host; `salt` starts at 1 and advances by the launch's unit count; no clear at allocation, none at the wrap |
The constraint of the night (coordinator, 5 October 2026): the whole working set on an 8 GB card stays under 6 GB
(1 GiB table, the layer 5 hot table, the scratch of every resident warp, buffers), which caps the scratch at tens of
KiB per warp. On an RTX 5090 at full occupancy (170 SMs x 64 warps = 10,880 warps, approximate hardware maximum;
the measured version 2 kernel ran 24 warps per SM, 4,080 warps, `docs/bench-log.md` M11, 4 October 2026):
| Scratch per warp | 10,880 warps | 4,080 warps (measured occupancy) | Table + scratch at 10,880 | Under 6 GB with a 1 GiB table |
|---|---|---|---|---|
| 32 KiB | 340 MiB | 128 MiB | 1,364 MiB | yes |
| 128 KiB | 1,360 MiB | 510 MiB | 2,384 MiB | yes |
| 1 MiB (the first experiment) | 10,880 MiB | 4,080 MiB | 11,904 MiB | no |
## 2. Question 1: uniformity of what is written
### 2.1 As functions
The fill of word j of slot s for lane nonce n is `splitmix32(((n ^ seed[j]) + s x 0x9e3779b1 + (j + 1) x 0x85ebca77))`.
`splitmix32` is a bijection of its 32-bit input; for fixed (seed, s, j) the input is a bijection of n. So over any
2^32 consecutive nonces every 32-bit value appears once as the fill of (s, j): uniform. Test
`fill_is_a_bijection_of_the_nonce`: 2^16 consecutive nonces give 2^16 distinct words for 7 slots x 3 word
positions; the fill of lane l at base b equals the fill of lane 0 at base b + l; it wraps with the nonce
(base 0xffffffe0, lane 32 equals nonce 0).
The rewrite `(x ^ w1, rotl(x, 7) ^ w2, x + w0)` is, for fixed old content w, a bijection of the fold value x in
EACH word. Test `rewrite_is_a_bijection_of_the_fold_value`: 2^16 consecutive x give 2^16 distinct words in each
position for 16 random w. Consequence: a uniform x gives a uniform word in every position, and the three words
are three images of the same x, so a rewritten slot carries exactly 32 bits of new state behind 96 bits of
storage (from w and any one written word, x is recovered; the test checks all three inversions).
The fold value x is `fold(dst, w)`, a bijection of `dst` for fixed w (xor, then rotate-multiply-xor twice; the
multiplier is odd). So the written words are uniform whenever `dst` is, and `dst` is a register of the running
program.
### 2.2 The attack: what a chip can precompute
The fill is a pure function of (seed, nonce, slot): precomputable, and meant to be (the verifier computes it too).
A chip never stores a fill; it computes it in about 10 integer operations when a slot is first touched. The written
words depend on `dst`, the register state at that instruction, which depends on every earlier instruction of the
hash, including the dataset loads. Nothing about them is precomputable before the hash runs. This is the whole of
what question 1 can give: the writes are as unpredictable as the registers. What that is worth is question 2.
### 2.3 The stats run (the `TESTS.md` section 3 shape)
Test `written_words_unbiased_and_rehit_rates`, M5 Max, 5 October 2026, `cargo test --test scratch`: for each
class, programs of `igneum-genesis`, `igneum-genesis/stats1`, `igneum-genesis/stats2`, 2^11 units each (196,608
hashes per class), closed-form dataset, every read-modify-write traced (`verify::interpret_warp_scratch`). Ones
count per bit of every written word and of the change each rewrite makes (written XOR read), sigma = sqrt(N)/2,
limit 6 sigma like the acceptance rule's output check.
| Class | Slots per lane | RMW per hash per lane | Rewrites traced | Max bias, written words (sigma) | Max bias, written XOR read (sigma) |
|---|---|---|---|---|---|
| scr2k32 | 64 | 16 | 3,145,728 | 2.61 | 3.40 |
| scr4k32 | 64 | 32 | 6,291,456 | 2.18 | 3.81 |
| scr8k32 | 64 | 64 | 12,582,912 | 3.63 | 2.25 |
| scr2k128 | 256 | 16 | 3,145,728 | 3.36 | 2.19 |
| scr4k128 | 256 | 32 | 6,291,456 | 2.71 | 3.68 |
| scr8k128 | 256 | 64 | 12,582,912 | 2.73 | 2.60 |
576 bit positions (6 classes x 3 words x 32 bits) at under 4 sigma is what fair coins give. Verdict: no structural
bias in what is written. Like `TESTS.md` section 3 this is a sanity check, not a proof of strength.
## 3. Question 2: no short cut avoids the writes
### 3.1 Inside a unit: the chain is dependent
Slot s of lane l, touched d times in a unit, holds `w_d = rewrite(x_d, w_{d-1})`, `w_0 = fill`, with
`x_i = fold(dst_i, w_{i-1})`. `x_i` depends on the slot content before it, which depends on every earlier fold
value of that slot; and `dst_i` is the register state, which the earlier fold values entered. Test
`slot_is_replayable_from_its_fold_values`: a slot after 64 read-modify-writes is reproduced from the fill and the
64 fold values; dropping one diverges. So a chip cannot skip a write and still read the slot later. It has three
ways to hold a slot, all exact:
| Store | Bytes per lane | Cost on a re-hit |
|---|---|---|
| Dense: every slot, 12 data bytes plus a valid bit | 12 x slots: 776 (64 slots), 3,104 (256), 24,832 (2,048) | one SRAM read |
| Sparse: only touched slots, 12 bytes plus a slot index | about 13 x distinct: 185 to 820 (table below) | one lookup |
| Implicit: only the fold values, 4 bytes plus a slot index per read-modify-write, replay on a re-hit | 5 x 8k: 80 (scr2), 160 (scr4), 320 (scr8) | d rewrites of 5 integer ops |
The implicit store is smaller than the dense one whenever `slots > 8k / 3`: at scr4 above 10.7 slots, at scr8
above 21.3. So "the smallest scratch at which keeping it implicitly is dearer than storing it" is 8k/3 slots per
lane, 2.7 to 5.3 KiB per warp at scr4 to scr8. Every size on the table, 32 KiB and above, is past it: a chip
keeps the scratch implicitly in 80 to 320 bytes per lane at any nominal size, and the replay cost is bounded by
the re-hit depth, which the next table measures.
### 3.2 The re-hit rate at 64 and 256 slots (and at 2,048)
Measured in the same test run (every read-modify-write of 196,608 hashes per class traced; a re-hit is a read of a
slot the same unit wrote earlier). Birthday: `distinct = S (1 - (1 - 1/S)^n)` for n uniform draws from S slots.
| Class | S | n = RMW per hash | Distinct slots, birthday | Re-hits, birthday | Re-hit %, birthday | Re-hit %, measured | Max chain depth seen | Slot histogram against uniform |
|---|---|---|---|---|---|---|---|---|
| scr2k32 | 64 | 16 | 14.26 | 1.74 | 10.9 | 12.58 | 7 | chi2 z 22,023; hottest slot 2.74x, coldest 0.83x |
| scr4k32 | 64 | 32 | 25.33 | 6.67 | 20.8 | 21.47 | 8 | z 10,880; 1.87x, 0.92x |
| scr8k32 | 64 | 64 | 40.64 | 23.36 | 36.5 | 36.99 | 9 | z 7,587; 1.39x, 0.91x |
| scr2k128 | 256 | 16 | 15.54 | 0.46 | 2.9 | 3.84 | 5 | z 19,146; 5.10x, 0.82x |
| scr4k128 | 256 | 32 | 30.14 | 1.86 | 5.8 | 6.25 | 6 | z 9,632; 3.06x, 0.90x |
| scr8k128 | 256 | 64 | 56.72 | 7.28 | 11.4 | 11.89 | 6 | z 5,873; 1.98x, 0.90x |
| 1 MiB (not run) | 2,048 | 32 | 31.76 | 0.24 | 0.8 | | | |
Two readings. First, the slot a read-modify-write addresses is the low 6 or 8 bits of a program register, and
those bits are not uniform: `or` sets them, `mul` clears them, so one slot of 256 is addressed 5.1 times as often
as the mean and the re-hit rate runs 2 to 33 percent above the birthday rate. For the dataset the same bias on the
low bits of a 28-bit address is harmless (it moves the read inside an item); for a 64-slot scratch it concentrates
the chain. Second, the chain depth is small: at scr4k32 the deepest slot in 196,608 hashes saw 8 earlier
read-modify-writes; a replay costs at most 8 x 5 integer operations, against about 1,170 for one dataset item.
### 3.3 The live state is bounded by the read-modify-write count, not by the size
The verifier evaluates one unit from nothing but (program, day, nonce group): `ScratchModel::new` per unit,
`verify.rs:296`. Every conforming GPU must therefore start every unit from the fill, which the tag does
(section 4). So no state crosses a unit boundary, and the state a unit can ever read back is what it wrote itself:
at most 8k slots per lane. The nominal size only sets how often those 8k writes land on the same slot (the table
above). The scratch's "memory" is 8k x 16 bytes per lane of touched slots, 256 bytes to 1 KiB at scr2 to scr8,
and a chip holds it implicitly in 80 to 320 bytes.
The attack of rolling back or sharing scratch between units has nothing to take: a unit starts from the fill
whatever ran before it, so a chip that clears 64 valid bits per unit has rolled back, and nothing one unit wrote
is readable by another. The CPU verifier is that chip.
### 3.4 The named chip, and what the scratch costs it
The strongest chip the plan has priced (coordinator, 5 October 2026): the whole 256 MiB cache on the die, computing
every dataset item on the fly. Its cache SRAM, from `docs/analysis/sram-mirror.md` revision 2 (`ca2-analysis`
e6085c6), headline at shipped-product density / bit-cell lower bound, dollars per good die approximate: 164 / 83
mm^2 and $30 / $13 at N7 (shipped density from AMD 3D V-Cache, 64 MB on 41 mm^2, Hot Chips 2021); 128 / 64 mm^2 and
$46 / $21 at N5, N3E and Intel 18A (TSMC N5 HD macro 31.8 Mib/mm^2 after assist overhead, SemiAnalysis, December
2022); 106 / 54 mm^2 and $56 / $26 at N2; with a 96 MB hot table 226 / 114 at N7, 175 / 89 at N5, 146 / 74 at N2.
The chip's cache cost in the table below is the N5 headline, 128 mm^2 and $46 per good die. It computes every item
through the mixer (`docs/analysis/m16-recompute-attacker-2026-10-05.md`: 128 items per hash, about 1,170 integer
operations per item, 150,000 per hash; at a 50 T op/s integer budget equal to a 5090's, approximate, 0.33 Ghash/s).
Against the measured version 2 rate of the RTX 5090, 139.7 MH/s (`docs/bench-log.md` M11, 4 October 2026), that is
2.4x before any fixed-function factor, 7x with the 3x the M16 analysis allows (approximate).
Units in flight on that chip. It has no DRAM latency to cover: every one of its 1,024 cache reads per hash is an
on-die SRAM read. Its hash latency is the dependent chain: 128 items x (8 dependent SRAM reads plus 9 mixer
applications). At about 10 ns per on-die read and about 40 ns per 130-operation mixer on a 16-wide integer
pipeline at 2 GHz (both approximate), an item is about 0.4 us and a hash about 50 us; at 0.33 Ghash/s that is
about 17,000 hashes in flight, 530 units of 32 lanes. A tighter pipeline halves it. The GPU covers DRAM latency (40 to 48 ns row
cycle, MEMSYS 2018, more under load) with 130,560 lanes in flight at the measured occupancy (4,080 warps x 32), 348,160 at
full occupancy, that is 8 to 20 times more lanes than the chip needs.
What the scratch costs that chip, per variant, with the arithmetic:
Chip cache mirror: 128 mm^2, $46 per good die (N5 headline; 64 mm^2, $21 bit-cell lower bound). Chip scratch SRAM at
the same two densities (2.1 MB/mm^2 headline, 4.2 MB/mm^2 lower bound at N5):
| Variant | Dataset loads per hash | Chip ops per hash | Chip rate at 50 T op/s | 5090 rate | Chip gain | Chip scratch SRAM at 17,000 lanes, implicit store | Same, dense 64-slot store | Dense store as mm^2, headline / lower bound (N5) | Share of the 256 MiB mirror (any density) |
|---|---|---|---|---|---|---|---|---|---|
| scr0 (control), 128 loads | 128 | 150,000 | 333 MH/s | 139.7 measured | 2.4x | 0 | 0 | 0 | 0 |
| 12.5% replaced (scr2) | 112 | 131,400 | 381 | 160 projected (128/112 x 139.7) | 2.4x | 1.4 MB | 13 MB | 6.2 / 3.1 mm^2 | 4.9% |
| 25% replaced (scr4) | 96 | 112,800 | 443 | 186 projected | 2.4x | 2.7 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 50% replaced (scr8) | 64 | 75,600 | 661 | 279 projected | 2.4x | 5.4 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 12.5% added (16 RMW beside 128 loads) | 128 | 150,200 | 333 | 139.7 or below | 2.4x or more | 1.4 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 25% added | 128 | 150,400 | 332 | 139.7 or below | 2.4x or more | 2.7 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 50% added | 128 | 150,800 | 332 | 139.7 or below | 2.4x or more | 5.4 MB | 13 MB | 6.2 / 3.1 | 4.9% |
| 256-slot dense store (128 KiB class), any share | | | | | | | 53 MB | 25 / 12.6 | 20% |
How the rows are computed: a read-modify-write costs the chip about 12 integer operations (fold and rewrite) and
one SRAM access; replacing a load removes an item derivation (1,170 operations); the 5090's rate for a replaced
load is projected from the measured distinct-load bound (the card's rate tracks distinct dataset loads per hash,
`docs/bench-log.md` 3 October, 23.7 G loads/s at 1 GiB; the readwidth agent's M5 Max measurement of the night,
relayed by the coordinator, shows the same: 27.7 MH/s at v2 to 29.4-31.7 at 25 percent replaced and 44.4-49.1 at 50
percent, 32 KiB per warp). The scratch SRAM is 17,000 lanes x 80 to 320 bytes (implicit) or x 776 bytes (dense at
64 slots) or x 3,104 bytes (dense at 256 slots); its share of the mirror is a ratio of bytes, 4.9 or 20 percent,
whichever density is used for both; the implicit store (the chip's cheaper choice at every size, section 3.1) is
0.5 to 2 percent. The chip's gain is set by operations per dataset item and the GPU's distinct-load bound, and the
scratch touches neither.
Plain answer to the coordinator's question: no read-modify-write share under the 6 GB cap, replaced or added,
brings the named chip under 2x. The share would be chosen as the smallest at which the chip falls under 1.5x, and
there is none: the gain is 2.4x at 0, 12.5, 25 and 50 percent, 32 or 128 KiB. This changes nothing about the public
claim that layer 3 would have changed: the claim must rest on the mixer, not on the scratch.
The lever that does move that chip, from the M16 table, beside it:
| Mixer cost multiplier | Chip ops per hash | Chip rate | Gain against 139.7 MH/s, no fixed-function factor | With a 3x factor (approximate) | CPU verify per warp (M16 table, scaled from 0.41 to 1.2 ms) | 5090 daily dataset build |
|---|---|---|---|---|---|---|
| x1 (today) | 150,000 | 333 MH/s | 2.4x | 7.2x | 0.4 to 1.2 ms | 13.4 ms |
| x2 | 300,000 | 167 | 1.2x | 3.6x | 0.8 to 2.4 ms | 27 ms |
| x4 | 600,000 | 83 | 0.6x | 1.8x | 1.6 to 4.8 ms | 54 ms |
| x8 | 1,200,000 | 42 | 0.3x | 0.9x | 3.3 to 9.6 ms | 107 ms |
The mixer multiplier leaves the honest hash rate untouched (the miner pays the mixer once a day), costs the chip
linearly, and is bounded by the 10 ms verification gate (x8 is at the gate's edge on this core, and the 2019-class
core of O-1.14 is unmeasured). The scratch costs the honest GPU a measured share of its rate when it spills the
cache and nothing when it does not, and costs the chip a few megabytes. The comparison is not close.
### 3.5 Where the GPU's writes would cost DRAM latency, and why that does not help
The GPU's hot scratch footprint is not the nominal size either: it is the slots in-flight units have touched,
about `warps x 32 lanes x distinct slots x 16 bytes` (x 2 at a 32-byte sector, approximate): on the 5090 at 4,080
resident warps and scr4, 25.3 slots at 64 or 30.1 at 256, 53 to 63 MB of slots, 100 to 125 MB in sectors, around
the card's 96 MiB L2 (`docs/bench-log.md`, 3 October). The readwidth agent's M5 Max rows (coordinator's message:
the rate rises with the share at 32 and 128 KiB) show the scratch sitting in that chip's caches at 4,096 warps.
To push the writes to DRAM latency the hot footprint must pass the last-level cache at the resident count:
`96 MiB / 4,080 warps = 24 KiB per warp`, which at 512 bytes of touched slots per lane per read-modify-write slot
means `8k x 512 B > 24 KiB`, k above 6 (above 48 read-modify-writes per hash) at ANY nominal size on the table, or
a higher resident count. That fits the 6 GB cap (it is the hot set, not the arena, that matters), and it costs
the honest miner a DRAM-latency read-modify-write per slot (a DRAM row cycle is 40 to 48 ns across DDR4, GDDR5 and
HBM2, Li, Reddy and Jacob, MEMSYS 2018; the loaded latency a GPU kernel sees is higher, approximate; DRAM latency
improved 1.3x in two decades while bandwidth improved 20x, Chang 2017, so no memory technology an attacker could
buy removes it, and no shipped mining chip has used HBM or stacked memory) while the named chip still keeps the same
hot set in a few megabytes of SRAM at 8 to 20 times fewer lanes in flight. The write path cannot be made to cost the
chip more than the GPU, because the GPU must keep 8 to 20 times more of it live.
## 4. Question 3: the verifier's one-warp simulation is exact
### 4.1 Lazy fill on both sides
The CPU initialises lazily with a written bit per (lane, slot), one model per unit. The GPU initialises lazily with
a 32-bit tag in word 0 of each 16-byte slot: a slot whose tag equals the unit's tag reads as written, any other
reads as the fill (`emit.rs:143-157`). There is no explicit fill and no reset between units of a persistent warp
(`emit.rs:159-175`: the loop over `g_` keeps the arena). The two agree if and only if, when a unit first touches a
slot, that slot does not already carry the unit's tag. That is:
1. Tags are unique over the life of the arena's contents (`tag = salt + g_`, `salt` the host's running counter).
2. The arena holds no word equal to a live tag in a slot's tag position before the unit writes it.
### 4.2 The attacks (the bug classes)
| Case | What happens | Today |
|---|---|---|
| Recycled allocation | A fresh process starts `salt` at 1 (`packbench.swift:144`, `host.c:1027`). If the driver hands back the previous process's arena with its contents (Metal, CUDA and OpenCL do not promise zeroed memory, approximate), slots tagged 1..N from the old run match the new run's first units exactly, and those units read stale words instead of the fill: a CPU mismatch on every colliding slot. | not guarded; passes on this Mac because fresh allocations read as zero in practice and tag 0 is never issued (luck, not contract) |
| Tag counter wrap | `salt` is 32 bits and advances by units per launch. A 5090 at 139.7 MH/s runs 4.37 M units/s, 2^32 units in 984 s: the counter wraps every 16.4 minutes on one card (81.8 minutes on the M5 Max at 28 MH/s). After the wrap a slot whose LAST writer carried the repeated tag reads as written. With 10,880 arenas each slot is rewritten about 395,000 times between two uses of one tag (at 64 slots a unit leaves a slot untouched with probability 0.60; 0.60^395,000 is 0), so on a full card the wrap is harmless in practice; on a one-warp launch repeated 2^32 times it is not. | not guarded |
| Tag 0 on zeroed memory | A host that starts `salt` at 0 gives unit 0 the tag 0, which a zeroed arena carries in every slot: unit 0 reads zeros for every first touch. | both hosts start at 1; nothing in the pack says they must |
| `groups` not a multiple of the warp count | Warps run different trip counts; the OpenCL local-memory exchange path carries a barrier inside the loop (spec 1.9), so a short warp hangs or desynchronises. | `packbench` refuses it; `host.c` rounds the batch |
### 4.3 The host contract that makes the simulation exact
A host of a scratch class MUST: allocate the arena as `warps x 32 x words_per_lane` words and zero it; issue tags
from a 32-bit counter that starts at 1 and advances by the unit count of every launch; zero the arena again before
any launch whose tags would pass 2^32 - 1 (tag 0 is never issued); launch `groups` as a multiple of the warp count.
The zeroing costs one memset of the arena (340 MiB at 32 KiB x 10,880 warps) every 2^32 units, 16 minutes on a
5090. This is the class fix for all four rows: with it the GPU's tag test and the CPU's written bit are the same
predicate.
### 4.4 The tests (Metal, M5 Max, 5 October 2026)
Two consecutive units on one persistent warp and the wrap inside a launch (`packbench --warps 1`,
`--batch-base 4294967040`, the option added on this branch); the hand-built edge programs that force every
read-modify-write of a hash onto one slot (so two consecutive units on one arena collide on every slot); the
deliberate breaks. Results in section 7.2. On the CPU, the same edge programs against an independent hand model
(a second interpreter with its own slot store, `tests/scratch.rs`): 56 of 56 cases match, and the hand model with
its rewrite words swapped mismatches on every case (the comparison has teeth).
## 5. Question 4: the attack surface of the writes
| Surface | Argument | Test |
|---|---|---|
| Out of bounds | `s_ = rN & (slots - 1)`, so `s_ < slots`; the lane's arena is `(warp_ x 32 + lane) x 4 x slots` words from the base, the access is `arena + 4 x s_ + 0..3`, the largest index is `warps x 32 x 4 x slots - 1`, the host's allocation. The emitter has one scratch template (`emit.rs:143`) and it masks. | `scr_packs_regenerate_and_pass_the_static_scratch_check`: 42 of 42 emitted kernels (7 scr packs x 6 files, the OpenCL bound file carrying two kernels) regenerate byte for byte from program.json and pass the text check: k masked slot definitions with the class mask, k tagged stores, 3k fill calls, one arena definition with the class stride, one tag definition, no `scratch[`; six deliberate breaks caught (section 8) |
| Aliasing between lanes | Lane-major: lane l of warp w owns words `[(32w + l) x 4S, (32w + l + 1) x 4S)`; two (w, l) pairs give disjoint ranges. Inside the range a slot is 4 words at `4 x s_`, so two slots of one lane are disjoint too. | the `lanevar` edge program: one init-dependent slot per lane, 32 lanes at 64 slots share slots in pairs by the birthday bound; any cross-lane aliasing would change the fold; 128 of 128 lanes on Metal (section 7.2) |
| Wave64 (two logical units in one hardware wave) | `warp_ = tid >> 5`, so the two halves get `warp_ = 2w` and `2w + 1`, two arenas; `gbase` and `tag` are per `g_`, per half. | not run on wave64 hardware (the OpenCL emulator's persistent launch is on the readwidth commit; unverified here) |
| Determinism: alignment | A slot is 16 bytes at byte offset `16 x (lane_base + s_)`; the arena base is the buffer base: Metal, CUDA and OpenCL allocations are at least 128-byte aligned (CUDA 256, OpenCL `CL_DEVICE_MEM_BASE_ADDR_ALIGN` at least the largest built-in type, approximate from memory), so every 16-byte vector access is aligned. | Metal: every run of section 7 |
| Determinism: ordering | A lane's two read-modify-writes of the same slot in one hash are a load and a store, then a load and a store, from one thread to one address: program order within a thread holds in every model. No other thread touches the slot (aliasing row), so no atomics, fences or barriers are needed and none are emitted. | `slot0` and `sixteen` edge programs: 64 and 128 dependent read-modify-writes on one slot per lane per hash, standalone and as the second unit on a warp |
| Determinism: vendors | The statement is integer only: xor, rotate by immediate, multiply, add, a 16-byte load and store. Bit-exact across Metal, CUDA and OpenCL by construction; measured only on Metal here. | Metal; CUDA and OpenCL runs are PC jobs (not mine tonight) |
| 32-bit nonce wrap | `gbase = baseNonce + g_ x 32` and `nonce = baseNonce + gid` wrap in 32-bit arithmetic; `scr_fill(gbase + lane)` wraps like the CPU's `base.wrapping_add(lane)`; `out[gid]` indexes by launch position, not by nonce. An aligned unit never straddles 2^32 (spec 1.9), so the wrap case is a launch whose unit SEQUENCE crosses it. | `packbench --batch-base 4294967040 --batch-log2 9`: 16 units from 0xffffff00, the ninth at gbase 0; fingerprint identical at 1 and 4 warps (section 7.2); every fuzz pack runs that launch |
## 6. Question 5: what a vector for the scratch class must carry
Before a scratch pack can be a conformance vector (plan step 4, "only then a vector"), it must carry, beyond what
`igneum-program-pack-3` carries today:
1. The class in the program id and the pack (`scr<k>k<kb>`: it is, `program_id_class`, `generator.rs:400-412`)
and the geometry (slots per lane, words per lane, bytes per warp: it is, `program.h`).
2. The fill and the rewrite as text (it is, `program.json` "scratch").
3. The host contract of section 4.3 as text in `program.h` and `program.json`: tag counter from 1, zero at
allocation and at the wrap, `groups` a multiple of the warp count. Not there today.
4. Vectors that exercise the tag path, which the three standard units do not reliably: two consecutive units on
one warp (bases 0 and 32 in one one-warp launch) for a program whose consecutive units collide on a slot. At
64 slots any generated program collides (25 touched of 64 per unit; the broken-tag run of section 8 was caught by
base 1,000,000, a warp's 16th unit, and NOT by a two-unit launch whose vectors lack base 32). At 2,048 slots two
consecutive units share a touched slot with probability about 0.4 (32 x 32 / 2,048 expected overlaps = 0.5), so
the standard vectors would miss a broken tag at the 1 MiB size with probability about 0.6 per unit pair. The
edge programs `slot0` and `sixteen` collide on every slot at every size: a vector set should carry one.
5. A unit in the top 256 nonces with the launch crossing 2^32 (`--batch-base` near the top, at least two warps).
6. The batch fingerprint declared independent of the warp count (`8c07620f4d9adefd` for scr4k32 at 2^12 nonces
from base 0 at 1, 2 and 128 warps, section 7.2): unit independence is the property the per-unit reset gives, and
a fingerprint that moved with the warp count would mean a unit read another unit's slot.
## 7. Tests and results
### 7.1 CPU (`igneum-pow/tests/scratch.rs`, `cargo test -j4 --test scratch`, M5 Max, 5 October 2026, 3.6 s)
| Test | What | Result |
|---|---|---|
| `rewrite_is_a_bijection_of_the_fold_value` | 16 random slot contents x 2^16 consecutive fold values, each written word distinct; the three inversions | pass |
| `fill_is_a_bijection_of_the_nonce` | 7 slots x 3 words x 2^16 nonces distinct; lane and base interchange; wrap | pass |
| `written_words_unbiased_and_rehit_rates` | 6 classes x 3 seeds x 2^11 units, every rewrite traced: bias within 6 sigma (worst 3.63), re-hit rate within 0.9x to 2x of birthday, slot histogram, depth histogram | pass (tables of sections 2.3 and 3.2) |
| `edge_programs_match_the_hand_model` | 7 edge programs x 2 geometries x 4 bases (0, 32, 0x7ffffff0, 0xffffffe0) against an independent hand model; the slots driven and the re-hit counts as built; the mutated hand model mismatches | 56 of 56 pass, 56 of 56 teeth |
| `scr_packs_regenerate_and_pass_the_static_scratch_check` | 7 scr packs: program and program id from program.json, 6 kernel texts byte for byte, static scratch check on all 42, the pack's vectors from the CPU; six deliberate breaks caught | pass |
| `fuzz_scr_programs_cpu` | 200 generated programs over the six classes, generator contract and acceptance on every one, 4 units each (one in 0..224, one around 2^31, one in the top 256 nonces, one uniform), traced run equal to the untraced run, every slot inside the lane; writes the 214 packs for Metal with `IGNEUM_SCRATCH_PACKS_OUT` | pass; 200 of 200 have a unit in the top 256 |
| `slot_is_replayable_from_its_fold_values` | 64 dependent read-modify-writes replayed from the fill and the fold values; one dropped diverges | pass |
The rest of the crate: 33 of 34 lib tests and all pack tests pass; `verify::tests::fold_and_wide_fetch` fails on the
readwidth tip itself (`verify.rs:508`, `k as u32 * 0x9E37_79B1` overflows under the test profile's overflow
checks; the readwidth agent's test, reported to its owner, not touched here).
### 7.2 Metal (`proto-metal/packbench` built from this branch, M5 Max, 5 October 2026, under `with-lock.sh run`)
| Run | Launch | Expected | Result |
|---|---|---|---|
| scr4k32, standard pack | 2,048 warps, 2^24 nonces, 1 batch | 3 of 3 standalone, 3 of 3 in batch | PASS, fingerprint `3d1af881bd978fb9`; 1.8 s wall for the whole run (compile, cache, 1 GiB build, vectors, batch) |
| scr4k32, warp-count independence | 2^12 nonces (128 units) at 1, 2 and 128 warps | one fingerprint | `8c07620f4d9adefd` at all three, PASS |
| scr4k32, wrap inside the launch | 512 nonces from 0xffffff00 at 1 and 4 warps | one fingerprint, the base-0 vector inside the window after the wrap | `8e9e233234d3a297` at both, in-batch 1 of 1, PASS |
| scr4k32, broken tag (`tag = salt`), standard vectors | 2,048 warps, 2^24 | the base-1,000,000 vector (warp 530's 16th unit) fails | standalone 3 of 3, in batch 2 of 3, overall FAIL (caught) |
| scr4k32, broken tag, two units on one warp | 1 warp, 2^6 | nothing to catch it: the standard vectors have no base 32 | standalone 3 of 3, in batch 1 of 1, PASS (missed: the point of section 6 item 4) |
| 14 edge packs (7 programs x 32 and 128 KiB), run A | 1 warp, 2^6 (units at bases 0 and 32 on one arena) | 4 of 4 standalone, 2 of 2 in batch each | 14 of 14 PASS (56 of 56 standalone units, 28 of 28 in batch) |
| 14 edge packs, run B | 1 warp, 2^9 from 0xffffff00 (16 units on one arena, the wrap inside) | 4 of 4 standalone, 3 of 3 in batch each | 14 of 14 PASS (56 of 56, 42 of 42) |
| edge `slot0` at 32 and 128 KiB, broken tag (`tag = salt`) | 1 warp, 2^6 | standalone 4 of 4, in batch 1 of 2, FAIL | as expected at both geometries: the second unit read the first's slot 0 and FAILED; the standalone units passed |
| edge `slot0` at 32 KiB, broken lazy fill (`m_` forced to all ones: a first touch reads the stale words) | 1 warp, 2^6 | standalone fails | 0 of 4 standalone, 0 of 2 in batch, FAIL (lane 0 of base 0: GPU `64b49aeb987dae69`, expected `9ff3a2021f66b5be`) |
| 200 fuzz packs (scr2k32 29, scr4k32 26, scr8k32 42, scr2k128 32, scr4k128 26, scr8k128 45; datasets 64 MiB, 256 MiB, 1 GiB) | 2 warps, 2^9 from 0xffffff00 (8 units per warp, the wrap inside) | 4 of 4 standalone, 2 of 2 in the window, 200 of 200 PASS | 200 of 200 PASS: 800 of 800 standalone units (25,600 hashes), 400 of 400 in batch; 91 s for the 200 runs |
Totals on Metal: 228 of 228 runs PASS where a pass was expected, 3 of 3 FAIL where a failure was built in.
## 8. Deliberate breaks (the watcher rule)
| Break | Where | Caught by | Evidence |
|---|---|---|---|
| One slot mask dropped (Metal) | copy of scr4k32 `program.metal` | static check: "masked slot followed by the load: 3, expected 4" | test output |
| Mask 63 changed to 127 on every RMW (Metal) | same | "masked slot followed by the load: 0, expected 4" | test output |
| Arena stride 256 changed to 128 words (Metal) | same | "arena definition: 0, expected 1" | test output |
| A stray `arena[0]` and `scratch[1]` access (Metal) | same | "arena mentions: 10, expected 9; direct scratch indexing: 1, expected 0" | test output |
| One slot mask dropped (OpenCL, CUDA) | copies of scr4k32 `kernel.cl`, `kernel.cu` | "masked slot followed by the load: 3, expected 4" | test output |
| Wrong class geometry or RMW count or kernel count passed against a right text | the same text | the check fails | test output |
| `tag = salt` (every unit of a launch shares the tag) | copy of scr4k32 `program.metal`, on the GPU | the base-1,000,000 vector in a 2,048-warp batch | `vectors standalone 3/3, in batch 2/3`, overall FAIL |
| the same on the `slot0` edge pack, two units on one warp | on the GPU | in-batch 1 of 2 | bench-log entry |
| lazy fill broken (`m_` all ones) | copy of the `slot0` edge pack, on the GPU | standalone vectors | bench-log entry |
| The hand model's rewrite words swapped | `tests/scratch.rs` | every edge case mismatches | 56 of 56 |
The out-of-bounds break (mask dropped) was not run on the GPU on purpose: Metal does not bounds-check device
buffers (`TESTS.md` section 5), so a run would read another lane's or another buffer's words and "did not crash" would
prove nothing. The static check is the guard, as it is for the dataset mask.
## 9. What is unverified
1. CUDA and OpenCL runs of the scratch packs on NVIDIA and AMD (PC jobs, reserved for the readwidth agent tonight);
the 5090's rate per variant, so the "projected" column of section 3.4 is the distinct-load bound, not a
measurement. Wave64 hardware for the two-arena argument.
2. The chip-side latency figures of section 3.4 (10 ns SRAM read, 40 ns mixer) are approximate; the conclusion
does not depend on them: at ten times the in-flight count the scratch is still under a sixth of the mirror.
3. The recycled-allocation case was not reproduced (it needs a driver that hands back live contents); the argument
is that nothing forbids it and the contract of 4.3 removes it.
4. The slot-bias finding (section 3.2) was measured on three seeds per class; the hottest-slot ratio will vary by
program.
5. `verify::tests::fold_and_wide_fetch` on the readwidth tip (section 7.1).
## 10. Recommendation
1. Layer 3 is sound as a construct: the written words are uniform, the chain inside a unit has no short cut, the
kernels cannot write out of bounds, and with the host contract of section 4.3 the CPU's one-warp simulation is
exact (14 edge packs, 200 fuzz packs, the wrap, consecutive units on one arena, on Metal).
2. Layer 3 is not sound as a chip-resistance layer, at the capped size or at any size: CPU verification resets the
scratch per unit, so its live state is 8k slots per lane whatever the arena, a chip keeps it implicitly in 80 to
320 bytes per lane, and the named chip (on-die cache mirror plus recompute) keeps its whole scratch in 1.4 to
13 MB of SRAM at 530 units in flight, 3 to 5 percent of its mirror. Its gain stays at 2.4x (7x with a 3x
fixed-function factor, approximate) at 0, 12.5, 25 and 50 percent, replaced or added, 32 or 128 KiB. No share
under the 6 GB cap brings it under 2x.
3. Taking the read-modify-writes from the 16 dataset loads makes the hash less memory-hard for everyone: the GPU
measured faster at every share on the M5 Max (readwidth rows), and the chip's operations per hash fall with the
loads. If a scratch is kept at all, add it beside the 128 loads. There is no reason found here to keep one.
4. The lever that moves the named chip is the M16 mixer multiplier: x2 to 1.2x, x4 to 0.6x against the measured
5090 rate, at 0.8 to 4.8 ms of verification per warp against the 10 ms gate. Decision 2 should price that
against the gate on the 2019-class core (O-1.14) rather than layer 3.
5. If the project lead keeps layer 3 for a reason outside this analysis: ship the host contract in the pack, add the four
vector items of section 6 (consecutive units with a forced collision, the wrap launch, the warp-count-independent
fingerprint, the contract text), and run the CUDA and OpenCL twins of section 7.2 on the PCs before the class
becomes a genesis rule.

View file

@ -0,0 +1,283 @@
# Layer 6: the SRAM mirror of the cache against published SRAM density, year 0 to 10
5 October 2026 (night), Counter ASIC 2.0 (`docs/plans/counter-asic-2.md`, layer 6), branch `ca2-analysis`. Every figure
below is either cited (paper, vendor document, URL, date) or labelled approximate. Nothing here is a measurement of a
chip. Numbers in this file were computed with the arithmetic shown; the script is in section 10.
Revision 2 (same night): the first draft priced the mirror from bit-cell area times a 0.70 array factor. The
coordinator's chip-economics research (sources below) showed that shipped cache-only dies land at about half that
density once assist circuits, redundancy, TSVs, power and test are in. Every table now carries two columns: the
shipped-product density as the headline and the bit-cell figure as the lower bound. The conclusion did not move; the
cost per die rose 2 to 3x.
## 1. The question
The lottery hash derives every dataset item from a 256 MiB cache (spec 01 sections 1.5 and 1.8). A chip that holds the
cache in on-die SRAM can recompute items instead of reading the dataset (ledger M16, the recompute attacker). Layer 6
asks whether the cache size, as the specification schedules it, keeps that SRAM mirror unaffordable for ten years of
the genesis schedule, and if not what growth rule would.
Two things also sit in a chip's SRAM budget if it mirrors the full read-only working set: the layer 5 hot table (32,
64 or 96 MB, a class parameter on `readwidth` b970dda, coordinator's note of 5 October) beside the 256 MiB cache. The
per-warp scratch of layer 3 (32 or 128 KB per warp, written, not read-only) is not mirrorable and is left out of the
mirror; it is counted in the 6 GB working-set budget in section 7.
## 2. What the specification schedules for the cache
| Quantity | Rule | Where |
|---|---|---|
| Dataset | 2 GiB at genesis plus 0.5 GiB per year (`N_d` grows about 23 KiB per day) | spec 01 section 1.13.3, Designed |
| Cache | 256 MiB, "prototype value, to be fixed at gate 1"; the rule that fixes it: "the cache must exceed the largest on-chip cache of any card that mines, and 96 MiB of L2 on the 5090 is the figure to beat" | spec 01 sections 1.5 and 1.16 |
| Cache growth | None. No section of `docs/spec/` grows the cache (grep of `docs/spec` for cache growth, schedule, doubling: only the dataset rule of 1.13.3 and the README's "growth" word, which refers to it) | this analysis, 5 October 2026 |
So the plan's layer 6 row ("already in the design; confirm the schedule") is half right: dataset growth is in the
design, cache growth is not. The cache is flat at 256 MiB for every year of the schedule as the spec stands. M16's
closing line names the rule the cache should get ("exceeds what one die can hold, and grows") as a gate 1 decision
that has not been taken.
## 3. SRAM density, cited: bit cells per node and shipped cache dies
### 3.1 Bit cells
| Node (vendor) | HD 6T bit cell, um^2 | Raw density, Mbit/mm^2 (1/cell) | Year of volume (approximate) | Source |
|---|---|---|---|---|
| N7 (TSMC) | 0.027 | 37.0 | 2018 | WikiChip, "TSMC Details 5 nm" (ISSCC/IEDM disclosures), https://fuse.wikichip.org/news/3398/tsmc-details-5-nm/ |
| N5 (TSMC) | 0.021 | 47.6 | 2020 | same (two N5 cells: HD 0.021, HP 0.025) |
| N3B (TSMC) | 0.0199 | 50.3 | 2022 to 2023 | WikiChip, "IEDM 2022: Did We Just Witness The Death Of SRAM?", https://fuse.wikichip.org/news/7343/iedm-2022-did-we-just-witness-the-death-of-sram/ (TSMC's IEDM 2022 N3 paper) |
| N3E (TSMC) | 0.021 | 47.6 | 2023 | same; Tom's Hardware, "TSMC's 3nm Node: No SRAM Scaling", https://www.tomshardware.com/news/no-sram-scaling-implies-on-more-expensive-cpus-and-gpus |
| N2 (TSMC) | 0.0175 | 57.1 | 2025 to 2026 | TSMC at IEDM 2024, reported by Tom's Hardware, https://www.tomshardware.com/tech-industry/tsmc-shares-deep-dive-details-about-its-cutting-edge-2nm-process-node-at-iedm-2024-35-percent-less-power-or-15-percent-more-performance ; ISSCC 2025 paper "A 38.1Mb/mm2 SRAM in a 2nm-CMOS-Nanosheet Technology", https://research.tsmc.com/page/memory/4.html |
| Intel 18A | 0.021 | 47.6 | 2025 to 2026 | ISSCC 2025 paper 29.2, "A 0.021 um^2 High-Density SRAM in Intel 18A RibbonFET Technology with PowerVia", https://www.researchgate.net/publication/389644177 ; IEEE Spectrum 26 Feb 2025, https://spectrum.ieee.org/sram-intel-tsmc |
| Samsung SF3 / SF2 | not disclosed as a bit cell area in anything found tonight (Samsung's ISSCC papers give assist circuits and macro figures, not the HD cell) | | | search of ISSCC 2021 to 2025 coverage, 5 October 2026; left out of the tables |
The stall. N3B's cell is 5% smaller than N5's and N3E's is the same size as N5's (0.021 um^2 both): zero SRAM
scaling from N5 to N3E (WikiChip IEDM 2022 article above; Tom's Hardware above; SemiAnalysis "TSMC's 3nm Conundrum",
https://newsletter.semianalysis.com/p/tsmcs-3nm-conundrum-does-it-even). N2's nanosheet cell recovers 17% (0.021 to
0.0175 um^2). So across 2020 to 2026 the HD bit cell shrank once, by 17%.
Macro density from the bit cell. WikiChip's and SemiAnalysis's convention is bit-cell density times about 0.70 for
the assist and periphery overhead (SemiAnalysis, December 2022: TSMC N5 HD SRAM macro 31.8 Mib/mm^2 after about 30%
assist overhead; WikiChip's 31.8 Mib/mm^2 for the 0.021 um^2 cell is the same arithmetic). The two ISSCC 2025 macros
bracket it: TSMC N2 38.1 Mb/mm^2 at a 0.0175 um^2 cell is 67%; Intel 18A 38.1 Mb/mm^2 array density and 34.3 Mb/mm^2
for the volume macro at a 0.021 um^2 cell are 80% and 72%. That is a macro on a test chip. It is the LOWER BOUND on
die area, not the die.
### 3.2 Shipped cache dies (what a whole die of SRAM really holds)
| Product | SRAM | Die | Node | MB per mm^2 | Source |
|---|---|---|---|---|---|
| AMD 3D V-Cache (Zen 3 SRAM chiplet) | 64 MB | 41 mm^2 | TSMC 7 nm | 1.56 | AMD at Hot Chips 33, reported by Tom's Hardware, August 2021, https://www.tomshardware.com/news/amd-unveils-more-ryzen-3d-packaging-and-v-cache-details-at-hot-chips ("the 3D V-Cache SRAM measures 41 mm^2", "64 MB of 7 nm SRAM"); the densest cache-only die that has shipped |
| Graphcore GC200 (with compute) | 900 MB | 823 mm^2 | 7 nm | 1.09 | coordinator's chip-economics research, 5 October 2026 (vendor figures) |
| Groq TSP | 220 MB | 725 mm^2 | 14 nm | 0.30 | same |
The V-Cache die is a pure SRAM die with its TSVs, redundancy, test and power: 1.56 MB/mm^2 at N7 against the bit-cell
figure 37.0 Mbit/mm^2 = 4.6 MB/mm^2 and the 0.70-macro figure 3.2 MB/mm^2. The shipped die is 0.48 of the macro
figure. The headline column below scales the V-Cache density to other nodes by the bit-cell ratio (0.027 / cell), an
approximation that assumes the periphery and TSV overheads scale with the cell, which they do not fully (so the
headline column is itself slightly optimistic for the attacker at N5 and below).
### 3.3 GPU on-die SRAM, the reticle, wafer prices
GPU on-die SRAM for scale: the RTX 5090 carries 96 MB of L2 (98,304 KB) on a 750 mm^2 TSMC 4N die with 92.2 billion
transistors; the full GB202 has 128 MB; the RTX 4090 had 72 MB and the RTX 3090 6 MB (NVIDIA, "RTX Blackwell GPU
Architecture" whitepaper v1.1, appendix table "L2 Cache Size", https://images.nvidia.com/aem-dam/Solutions/geforce/blackwell/nvidia-rtx-blackwell-gpu-architecture.pdf).
At the V-Cache density scaled to N5 (2.0 MB/mm^2) that L2 is about 48 mm^2 of the 750 (6%), approximate. The
RX 9070 XT carries 64 MB of Infinity Cache plus 8 MB of L2 (vendor figures, approximate, bench-log "the 9070 XT on the
eGPU").
Reticle: the EUV field is 26 x 33 mm = 858 mm^2, about 830 mm^2 usable after scribe lanes (SemiAnalysis, "Die Size
And Reticle Conundrum", https://newsletter.semianalysis.com/p/die-size-and-reticle-conundrum-cost ; WikiChip "Mask",
https://en.wikichip.org/wiki/mask). The 5090's 750 mm^2 is 90% of it.
Wafer prices (approximate; TSMC publishes none, every figure is supply-chain reporting): N7 about $9,500, N5 and N3
about $20,000 (Silicon Analysts, "Wafer Pricing by Node", September 2026, https://siliconanalysts.com/data/wafer-pricing);
N2 about $30,000 (Tom's Hardware, https://www.tomshardware.com/tech-industry/semiconductors/tsmc-could-charge-up-to-usd45-000-for-1-6nm-wafers-rumors-allege-a-50-percent-increase-in-pricing-over-prior-gen-wafers).
## 4. Die area to mirror the cache, per node, two columns
Headline = V-Cache density (41 mm^2 per 64 MiB at N7) scaled by the bit-cell ratio. Lower bound = bits / (raw
density x 0.70). Columns: the 256 MiB cache alone, the cache plus the 96 MB hot table of layer 5 (as MiB), and the
larger caches of the options in section 7. Area in mm^2; a figure over 830 is split into the dies shown.
| Node | 256 MiB, headline | 256 MiB, lower bound | 256 + 96, headline | 256 + 96, lower bound | 512 MiB, headline / lower | 1 GiB, headline / lower | 4 GiB, headline / lower |
|---|---|---|---|---|---|---|---|
| N7 | 164 | 83 | 226 | 114 | 328 / 166 | 656 / 331 | 2,624 (4 dies) / 1,325 (2 dies) |
| N5 | 128 | 64 | 175 | 89 | 255 / 129 | 510 / 258 | 2,041 (3 dies) / 1,031 (2 dies) |
| N3B | 121 | 61 | 166 | 84 | 242 / 122 | 483 / 244 | 1,934 (3 dies) / 977 (2 dies) |
| N3E, Intel 18A | 128 | 64 | 175 | 89 | 255 / 129 | 510 / 258 | 2,041 (3 dies) / 1,031 (2 dies) |
| N2 | 106 | 54 | 146 | 74 | 213 / 107 | 425 / 215 | 1,701 (3 dies) / 859 (2 dies) |
One reticle (830 mm^2) holds, at the headline density, 1.3 GiB of SRAM at N7, 1.6 GiB at N5, N3E and 18A, 1.9 GiB at
N2 (lower-bound column: 2.5, 3.2, 3.9 GiB).
Against the figures the ledger carries: M16's "100 to 300 mm^2" (low end from a 0.02 um^2 cell with overhead, high
end from wafer-scale parts at about 1 MB per mm^2) brackets the headline 106 to 164 mm^2 well; the plan's "about
45 mm^2 at a leading node" is below even the lower bound and should be read as the bit-cell area with no overhead.
The right figures for the ledger are 106 to 164 mm^2 (shipped density) with 54 to 83 mm^2 as the floor.
## 5. Cost per good die, two columns
Dies per 300 mm wafer by the usual approximation pi x 150^2 / A minus the edge term pi x 300 / sqrt(2A); yield by
Poisson exp(-A x D0) with D0 = 0.1 defects per cm^2 (an assumption, approximate; SRAM arrays carry redundancy so
real yield is higher, which lowers these costs). Cost per good die = wafer price / (dies x yield). Packaging, test,
the logic beside the SRAM and the design (masks at N5 and below run into the tens of millions of dollars,
approximate) are not in these numbers; they are per-die silicon only. Headline / lower bound in each cell.
| Node, wafer price | 256 MiB | 256 + 96 MiB | 1 GiB | 4 GiB |
|---|---|---|---|---|
| N7, $9,500 | 164 mm^2, 379 dies, yield 0.85: $30 / $13 | $44 / $19 | $224 / $75 | $896 (4 dies) / $456 (2 dies) |
| N5, $20,000 | 128 mm^2, 495 dies, 0.88: $46 / $21 | $68 / $30 | $306 / $111 | $1,512 (3 dies) / $621 (2 dies) |
| N3B, $20,000 | 121 mm^2, 524 dies, 0.89: $43 / $20 | $63 / $28 | $280 / $103 | $1,371 (3 dies) / $569 (2 dies) |
| N3E, 18A, $20,000 | $46 / $21 | $68 / $30 | $306 / $111 | $1,512 / $621 |
| N2, $30,000 | 106 mm^2, 600 dies, 0.90: $56 / $26 | $81 / $37 | $343 / $131 | $1,641 (3 dies) / $696 (2 dies) |
Reading. The silicon for a 256 MiB mirror is $30 to $56 per die at shipped density (2 to 3x the first draft's
figure), under $90 with the hot table. A funded chip programme pays that without noticing: it was never the SRAM
that priced the recompute attacker out, and the plan's premise for layer 6 ("the SRAM mirror stays unaffordable")
does not hold for the cache as a mirror and did not hold at genesis either. A 1 GiB cache is a 425 to 656 mm^2 die
($224 to $343), affordable too; 4 GiB is a 3 to 4 die part at about $900 to $1,600 of silicon, which is a different
product but not an impossible one (the attacker's problem at that size is the 1,024 dependent cross-die reads per
hash, section 6).
## 6. What the mirror buys the attacker, year by year
From M16 (`docs/analysis/m16-recompute-attacker-2026-10-05.md`): with the cache on die the attacker recomputes 128
items per hash at about 1,170 integer operations and 8 dependent 64-byte cache reads each, about 150,000 operations
and 1,024 dependent SRAM reads per hash. At a 5090-class integer budget (about 50 T op/s, approximate) that is
0.33 Ghash/s against the honest 141 Mhash/s projected for version 2 programs: 2.4x at equal silicon before any
fixed-function factor, 3x to 6x with one (approximate). The SRAM is 106 to 164 mm^2 of that chip at the headline
density (14 to 22% of a 750 mm^2 die; the m16 model's 13 to 40% band holds), so the mirror is cheap and the recompute
route is bound by integer throughput, not by SRAM.
The layer 5 hot table changes nothing in that arithmetic: the hot table is read-only and derived from the day key
like the cache, so a chip mirrors it in the same SRAM (another 32 to 96 MB, 24 to 48 mm^2 at N5 headline) and reads
it at SRAM latency, which is exactly what a GPU's L2 does with it. Layer 5 taxes the DRAM-only chip (the one without
SRAM); it does not tax the SRAM chip.
Dataset growth does not touch the recompute attacker: the attacker never holds the dataset. It taxes the
partial-store attacker (O-1.6, the time-memory curve, not drawn) and the honest card.
Year by year under the schedule as it stands (flat 256 MiB), the mirror's area at the best node available that
year, headline density. Node years are approximate; the density trend from 2018 to 2025 is 37.0 to 57.1 Mbit/mm^2
raw, 1.54x in 7 years, about 6% per year, and it came in one step (N2); the extrapolation past 2026 assumes that
average holds (approximate, and optimistic for the attacker: A16 and A14 have no disclosed SRAM cell yet).
| Year | Calendar (approximate) | Dataset, GiB | Cache (spec) | Best node | Mirror of the cache, headline (lower bound), mm^2 | With a 96 MiB hot table, headline, mm^2 | Mirror as a share of a 750 mm^2 die |
|---|---|---|---|---|---|---|---|
| 0 | 2027 | 2.0 | 256 MiB | N2 (cited) | 106 (54) | 146 | 14% |
| 1 | 2028 | 2.5 | 256 MiB | N2 or A16 | 103 (52) | 142 | 14% |
| 2 | 2029 | 3.0 | 256 MiB | trend | 95 (48) | 130 | 13% |
| 3 | 2030 | 3.5 | 256 MiB | trend | 89 (45) | 123 | 12% |
| 4 | 2031 | 4.0 | 256 MiB | trend | 84 (43) | 116 | 11% |
| 5 | 2032 | 4.5 | 256 MiB | trend | 79 (40) | 109 | 11% |
| 6 | 2033 | 5.0 | 256 MiB | trend | 75 (38) | 103 | 10% |
| 7 | 2034 | 5.5 | 256 MiB | trend | 71 (36) | 97 | 9% |
| 8 | 2035 | 6.0 | 256 MiB | trend | 67 (34) | 92 | 9% |
| 9 | 2036 | 6.5 | 256 MiB | trend | 63 (32) | 87 | 8% |
| 10 | 2037 | 7.0 | 256 MiB | trend | 59 (30) | 82 | 8% |
Reading. A flat cache's mirror shrinks from 14% to 8% of a large die over the decade, and a 5090-class consumer GPU
already carries 96 MB of L2 on one die with the full GB202 at 128 MB; at the 2020 to 2025 pace of GPU L2 growth
(6 MB, 72 MB, 96 MB on the three NVIDIA flagships in the whitepaper table) a consumer GPU could hold 256 MiB on die
within the decade. The spec's own rule for the cache ("must exceed the largest on-chip cache of any card that
mines") would then be broken by a flat cache. That is the real reason to grow it: not to price a chip out (section
5 shows the SRAM cannot do that) but to keep the cache out of every GPU's own cache, so the honest hash stays
DRAM-latency-bound and the recompute route stays a route only a custom chip can take.
## 7. Answer to the layer 6 question, and the options
Does the flat 256 MiB cache keep the SRAM mirror unaffordable through year 10? No. It is affordable at year 0 ($30 to
$56 of silicon per die at shipped density, section 5) and gets cheaper. What keeps the recompute attacker near 1x is
M16's integer arithmetic and the mixer-cost lever (4x the mixer cost puts the equal-silicon gain at 0.36x, bounded
by the CPU verify gate), not the cache size. The cache size does one other job, keeping the cache larger than any
GPU's L2, and that job needs growth.
Options for the cache rule, with the honest costs each implies. Verifier fill time is 0.2 s per 256 MiB on one core
(spec 1.12: "a 0.2 s CPU cache fill", from the measured 175 to 190 ms of section 1.8.3), scaled linearly; the
verifier holds the whole cache (section 1.11), so its memory is the cache size plus the program and the interpreter.
GPU fill: 0.67 ms per 256 MiB on the 5090 (section 1.8.3), linear. The GPU dataset build (13.4 ms per 1 GiB on the
5090, section 1.8.3) depends on the dataset size, not the cache size; a larger cache spreads the build's 8 dependent
reads per item over more memory, which on a GPU means more of them miss L2 and the build slows by some factor
between 1x and the L2-to-DRAM latency ratio, which is a measurement to take (approximate; owed). Mirror area is at N2
headline density (lower bound in brackets), the node of the first years; at the trend's year-10 density divide by
about 1.8.
| Option | Rule | Cache at year 0 / 4 / 10 | Mirror at N2, headline (lower bound), year 0 / 4 / 10, mm^2 | Dies at year 10 (830 mm^2 reticle), headline | Verifier fill, one core, year 0 / 10 | Verifier memory, year 10 | GPU cache fill (5090), year 10 | Keeps the cache above a 96 MB L2 at year 10 | Keeps it above a 256 MB L2 |
|---|---|---|---|---|---|---|---|---|---|
| A, as specified | flat 256 MiB | 256 / 256 / 256 MiB | 106 (54) / 106 / 106 | 1 | 0.2 / 0.2 s | 256 MiB | 0.7 ms | yes, 2.7x | no |
| B | cache = dataset / 8 (today's ratio) | 256 / 512 / 896 MiB | 106 (54) / 213 (107) / 372 (188) | 1 | 0.2 / 0.7 s | 896 MiB | 2.3 ms | yes, 9.3x | yes, 3.5x |
| C | cache doubles when the dataset doubles (the dataset's own clock: year 4, then year 12) | 256 / 512 / 512 MiB | 106 (54) / 213 (107) / 213 (107) | 1 | 0.2 / 0.4 s | 512 MiB | 1.3 ms | yes, 5.3x | yes, 2x |
| D | cache = dataset / 4 | 512 / 1,024 / 1,792 MiB | 213 (107) / 425 (215) / 744 (376) | 1 | 0.4 / 1.4 s | 1.75 GiB | 4.7 ms | yes | yes, 7x |
| E, one reticle | cache sized so the mirror exceeds one reticle at the node of the day: 2 GiB at N2 headline density (section 4; 4 GiB on the lower bound), growing with density | 2 GiB / about 2.3 / about 3.5 GiB | 850 / 850 / 850 (by construction) | 2 | 1.6 / 2.8 s | 3.5 GiB | 5.4 / 9.4 ms | yes | yes |
Where the working set enters (coordinator's budget: 1 GiB table + hot table + scratch for every resident warp +
buffers under 6 GB on an 8 GB card): the cache is not in the miner's working set at hash time (the dataset is built
from it once a day and the cache can be dropped or kept), so options A to D do not move that budget; the dataset's own
growth does (2 GiB at genesis, 4 GiB at year 4, 7 GiB at year 10, which is past an 8 GB card at about year 8 on its
own). Option E's 2 GiB cache would have to be built on the card and dropped, which is fine for a 16 GB card and tight
on an 8 GB one at build time (2 GiB cache + 2 GiB dataset + hot table). The per-warp scratch at 170 SMs x 64 warps
(approximate, readwidth) is 340 MB at 32 KB and 1.36 GB at 128 KB per warp; with the 1 GiB table, a 96 MB hot table
and buffers that is 1.5 to 2.5 GB at the prototype dataset size, 2.5 to 3.5 GB at the 2 GiB genesis size, inside
6 GB either way.
Recommendation. Option C (the cache doubles when the dataset doubles) is the one that keeps the spec's own rule true
with the smallest verifier cost: it ties the cache to a clock the spec already has, keeps `AND MASK` (a power of two
every step, which is the 1.13.3 option (b) argument again), costs the verifier 0.4 s and 512 MiB at year 4 and nothing
more until year 12, and keeps the cache 2x above a 256 MB GPU L2 if one appears. It does not price a chip out; nothing
about cache size does (section 5). The lever that does is the mixer cost multiplier of M16, which is the gate 1
decision to take beside this one. Option B is the same idea in a smooth form and costs the verifier 0.7 s at year 10.
Option E is the only one that makes the mirror a multi-die part and it costs every verifier 1.6 s and 2 GiB at
genesis (at the headline density; the lower-bound density would ask for 4 GiB and 3.2 s), which fails the spirit of
the 10 ms verify gate (the fill is once a day, but a light node joining pays it on every day it syncs across).
Decision for the project lead, at gate 1: A, B, C, D or E above, together with M16's mixer multiplier. Nothing here changes a
vector today: the cache size is a prototype value of spec 1.16 and the growth rule would be a new sentence in 1.13.3.
## 8. Why the latency bound is the property to lean on (citations behind the plan's rule)
The plan's "what stays true" paragraph says DRAM latency is the same physics for everyone and bandwidth per watt is
what a custom memory chip buys. The sources behind that:
| Claim | Figure | Source |
|---|---|---|
| Random-access DRAM latency is the same across memory types | Row cycle time 40 to 48 ns across DDR4, GDDR5 and HBM2 | Li, Reddy and Jacob, "A Performance and Power Comparison of Contemporary DRAM Architectures", MEMSYS 2018 (coordinator's chip-economics research, 5 October 2026) |
| Latency does not scale, bandwidth does | DRAM latency improved about 1.3x in two decades while bandwidth improved about 20x | K. Chang, "Understanding and Improving the Latency of DRAM-Based Memory Systems", PhD thesis, CMU, 2017 (same research) |
| No mining chip has bought latency with exotic memory | No shipped mining chip has used HBM or stacked memory; the Ethash chips used DDR3, GDDR6 and undisclosed types | same research; the Ethash chip gain of about 3x in the plan came from bandwidth per watt, not latency |
| The honest hash is latency-bound on every card measured | The hash runs within a few percent of 1/128 of each card's dependent random-read ceiling (5090, 9070 XT, M5 Max) | `docs/bench-log.md`, "the 9070 XT on the eGPU", 5 October 2026 (measured) |
Reading for layer 6: an SRAM mirror beats DRAM latency by about 10x per read (a 64 MiB buffer inside the 9070 XT's
Infinity Cache chased at 9.2 G loads/s against 2.5 in GDDR6, the same bench-log entry; the 5090's L2 at 5.8x the
hash rate of its 1 GiB dataset, M16), which is why the recompute attacker is bound by the 1,024 dependent SRAM reads
and the 150,000 integer operations per hash and not by the SRAM's size or price. The cache size decides whether the
mirror is one die or several (section 4); it does not decide whether the mirror exists.
## 9. What is cited, what is approximate, what is owed
| Item | Status |
|---|---|
| Bit cells for N7, N5, N3B, N3E, N2, Intel 18A | cited (section 3.1) |
| Shipped cache-die density (AMD V-Cache 64 MB on 41 mm^2 at 7 nm; Graphcore GC200; Groq TSP) | cited (section 3.2; V-Cache checked against Tom's Hardware's Hot Chips 33 report, 5 October 2026; the Graphcore and Groq rows are from the coordinator's research and were not re-checked tonight) |
| Scaling the V-Cache density to other nodes by the bit-cell ratio | approximate, stated |
| Samsung SF2 or SF3 bit cell | not found; left out |
| Array efficiency 0.70 | WikiChip's and SemiAnalysis's convention, bracketed by two ISSCC 2025 macros (67 to 80%); a macro figure, used only as the lower bound |
| Wafer prices | approximate, supply-chain reporting, cited |
| D0 = 0.1 per cm^2, Poisson yield | assumption, stated |
| Node years and the 6% per year density trend past 2026 | approximate, extrapolated from cited 2018 to 2025 points |
| GPU L2 sizes | cited (NVIDIA whitepaper); AMD Infinity Cache approximate |
| Latency citations (MEMSYS 2018, Chang 2017, mining-chip memory types) | from the coordinator's research, not re-read tonight |
| Recompute attacker arithmetic | M16, which is itself arithmetic on measured rates, not a chip measurement |
| Dataset-build slowdown at a larger cache on a GPU | owed, a measurement (5090 at a 512 MiB and 1 GiB cache) |
| The on-die emulation of M16 (inline kernel with a 64 MiB cache inside the 5090's L2) | still a PC job (M16) |
## 10. The arithmetic
```
MiB = 2^20; bits = cache_MiB * MiB * 8
headline_mm2 = cache_MiB * (41 / 64) * (cell_um2 / 0.027) (V-Cache: 41 mm2 per 64 MiB at N7, scaled by cell)
raw_Mbit_per_mm2 = 1 / cell_um2 (1e6 cells per mm2 per um2 of cell)
lower_bound_mm2 = bits / (raw * 0.70 * 1e6)
dies_per_wafer = pi * 150^2 / area - pi * 300 / sqrt(2 * area)
yield = exp(-area_mm2 * 0.001) (D0 = 0.1 per cm2)
cost_per_good_die = wafer_price / (dies * yield); over 830 mm2: k = ceil(area / 830) dies of area / k, cost x k
reticle_GiB = 830 / (mm2 per MiB) / 1024
```
Run on 5 October 2026 with Python 3 on the M5 Max; the printed tables are the ones above, rounded.

View file

@ -1451,6 +1451,32 @@ The adopted fee table (spec 05 section 5.11) reaches the devnet by `fees_v1_acti
Block rate for H: DAA 111,230 at 15:23Z, 112,227 at 15:40:13Z, 0.965 blocks/s; H = 210,000 is 24 h ahead of a publish before about 19:50Z on 5 October (the runbook moves it otherwise).
## 5 October 2026 (night), the C4 fix: certificate-driven reorg
Owner: the consensus engineer and cryptographer agent, worktrees `igneum-wt-c4` (branch `c4-fix`) and `vendor/igneum-node-c4` (fork branch `c4-fix` on release-0.3.6 a24ab01a). Harness `tools/finality-attacks/c4.mjs` on the fast-time 3-node network (100-ms proxied links), node built on the Mac in `vendor/igneum-node/target-c4` from the fork worktree (an APFS clone of `target-036`), suites on PC 2 through `tools/build-job.mjs`. The Mac carried two other builds and the M20 live sync throughout; every figure is a count, an index or a second from the harness clock.
**The cause, in the code.** `processes/finality.rs`: `ingest_certificate` verified a certificate only when its block was the node's own determination at that index (`cp.hash == cert.checkpoint`); any other block went to `hold_pending`, and nothing ever tried the pending certificate against the table at its own block. `fork_choice_lock` reads `state.locks`, which only `evaluate` filled, and `evaluate` only ever ran over the node's own determination. So a certified checkpoint off the node's chain never became a lock and never constrained the sink search, whatever spec 3.5 says. Second cause, found tonight on the harness: `protocol/flows/src/v10/blockrelay/flow.rs` skips a relayed block whose blue work is under the virtual's merge-depth root ("hence we are skipping it"), and the certified chain is lighter by construction, so the node on the heavier side never received the certified chain's blocks at all: in the first runs on the fixed consensus n0 held B's certificates by gossip for the whole heal window and B's blocks never arrived (n0's log shows only its own blocks "via submit block" after the reconnect).
**The fix.** Fork: `ingest_off_chain` (verifies against `voters_at` of the certificate's own block, Q3 and Q5 by `quorum_at` from that block's past, the lock chain by `off_lock_chain`, then LOCKED with a `FinalityLock` notification and a `VirtualStateProcessingMessage::Resolve` nudge so the sink moves without waiting for a block); `retry_pending_off_chain` on every virtual change; the lock-chain guard in `evaluate` (a determination off the chain through the node's nearest locks never locks and never aggregates); `fork_choice_lock` reports a lock beyond the depth-based finality point once; `wants_unknown_certified_block` and the relay-flow bypass of the merge-depth skip while a pending certificate names a block the node lacks (`finality_wants_blocks` through `ConsensusApi` and the session). Not gated on `finality_v3_activation_daa`: rule v2 took the same pending path.
**Unit tests (PC 2, job build-20261005-180827, 18:09 UTC): `kaspa-consensus` 97 passed, 0 failed, 3 ignored; `kaspa-consensus-core` 101 passed.** New: `a_lighter_certified_chain_wins_and_a_heavier_uncertified_one_does_not_override_it` (main chain 77 blocks locks to 13, a 7-block side chain's index-14 certificate is adopted, the sink moves to the side tip with no new block, ten more main-chain blocks do not move it back, a second certificate at 14 over the main block is CONFLICTING and the lock stands, the side chain then locks 15), `the_certificate_driven_reorg_holds_under_rule_v2` (the same at `finality_v3_activation_daa` never), `a_chain_that_misses_an_adopted_lock_never_locks_here` (the evaluate guard and the off-lock conflict). `reorg_past_an_unlocked_checkpoint_re_determines_it_and_verifies_the_pending_certificate` rewritten for the new behaviour (pending while the block is unknown, adopted when it arrives). `kaspa-p2p-flows` lib tests do not compile on release-0.3.6 before or after this change (nine `epoch_seed_headers` errors in the pruning-proof message tests; the M20 job build-20261005-172340 hit the same nine an hour earlier). Six PC 2 jobs were lost tonight to two tooling faults, both fixed in the class: `igneum-ota-sign embedded | head -1` under `pipefail` (SIGPIPE panic, four scripts, `tools/ci/signer-pipe-check.sh`) and the one shared `build-inputs.zip` in the downloads folder (a job published while another agent's pack landed pinned that agent's sources, three times; `build-job.mjs` now names every job's zip).
**Harness, weight against work (B four keys and 70% of the weight, A two keys and 30%; at the cut A mines 0.6 and B 0.4 blocks/s; `WINDOW` = weight window, ban and `min_daa` at fast time).** The 120-DAA window of the earlier runs turns every long split into F21's partition-longer-than-a-window shape once the p2p reconnect is added: n0 dials the proxy again on the connection manager's backoff, 84 to 114 s after the heal in every run tonight (the original sweep's 6 s was a short cut), so A's chain is 130 + 84 s = 128 DAA past the cut before any certificate can reach it, past its 120-DAA frozen table (v3) or its own two-thirds share of a sliding table (v2, 126 DAA at W 240 and s 0.3), and A locks alone first. With `WINDOW=240` and `WARM=320` the bound is 400 s (v3) or 210 s (v2) after the cut.
| Run | Node | Rule, W, split | B locks during the split | n0 reconnected | n0 adopted off-chain | Final chain | Conflicting | Disagreeing | Verdict |
|---|---|---|---|---|---|---|---|---|---|
| on 90 s (the sweep's framing) | c4 consensus fix, no sync hook | v3, 120, 90 s | 0 (36 blue blocks for B, a new index needs 50) | 6 s | 0 | A, all three (no certificate to follow) | 0 | 0 | not the C4 shape |
| v2 90 s | same | v2, 120, 90 s | 1 (index 8) | 6 s | 0 | apart | 5 / 5 / 5 | 2 | n0 locked 10 alone at 18:32:42, B's certificate for 8 reached it at 18:32:43: F21's bound (63 DAA of A's own chain) crossed before the heal |
| on 130 s | same | v3, 120, 130 s | 1 (index 9) | 84 s | 0 | apart | 9 / 3 / 3 | 2 | n0 locked 12 alone at DAA 359, one window after lock 8 at 239, 6 s before the reconnect |
| off 150 s (control) | same | no certificate, 150 s | 0 | 96 s | 0 | A (heavier), all three; B's nodes re-determined 2 indices | 0 | 0 | PASS, as in the sweep |
| on 130 s, W 240 | same | v3, 240, 130 s | 2 (10, 11) | 114 s | 0 (certificates 13 and 14 pending, blocks unknown) | apart | 0 | 0 | the sync gap: n0 never received a B block |
| v2 130 s, W 240 | same | v2, 240, 130 s | 2 (10, 11) | 114 s | 0 | apart, n0 locked 16 alone at 293 s | 0 / 1 / 1 | 0 | the sync gap again (n0 reconnected after v2's 210-s bound) |
| v2 130 s, W 240 | c4 fix with the sync hook | v2, 240, 130 s | 1 (index 12) | 84 s | 3 (12, 13, 14 within 2 s of the first B block; 11 re-determined) | B, all three, A's split tip abandoned | 0 | 0 | PASS |
| on 130 s, W 240 | same | v3, 240, 130 s | 0 (Poisson: 52 blue blocks, the index fell just short) | 84 s | n1 1, n2 2 (B's nodes adopted A's post-heal certificates and moved before IBD) | A, all three | 0 | 0 | the mirror case; not the C4 shape |
| on 140 s, W 240, addPeer at the heal | same | v3, 240, 140 s | 2 (11, 12, first at 12 s) | 3 s (the harness now dials through `addPeer`; the address goes as `{ip, port}`) | 1 (12 by certificate; 11 verified on the new chain) | B, all three, A's split tip abandoned | 0 | 0 | PASS |
Reading. With the consensus fix and the sync hook, a node on the heavier chain that receives a certificate for a chain it has never seen fetches that chain, verifies the certificate at its own block, locks it, moves its sink to the lighter certified chain and re-determines its own records onto it (the v2 W 240 row: 0 conflicts, 0 disagreements, every node on B's chain, which is the spec's F1 and the design's Fork choice items 1 to 4). The same holds under rule v3 with the frozen table on (the last row: B certified 11 and 12 during a 140-s split, n0 reconnected 3 s after the heal once the harness dialled through `addPeer`, adopted 12 by certificate and ended on B's chain with the other two, 0 conflicts, 0 disagreements). The fix does not and cannot cover a partition that outlasts the bound before the certificate arrives (rows 2, 3 and 6): there the node has already locked alone and 3.11.4 keeps that lock, the late certificate is CONFLICTING for the operator. On the live devnet (W 7,200 DAA, two hours) the bound is two hours after a side's last lock, so every partition under that heals by certificate. Raw: `scratchpad c4-results-*.md`, node logs `c4-*-n0.log`.
## 5 October 2026 (evening), FUD ledger sweep round 6
Owner: the consensus engineer and cryptographer agent, worktree `igneum-wt-fud-a` (branch `fud-a`), 15:45 to 16:40 UTC. The Mac was loaded throughout (two cargo builds, a txgen run and a fee-switch simnet by other agents; load average over 100), so every figure below is a count, an index, a byte or a number from another machine; the only millisecond figures are the browser verifier's, taken as ratios and labelled. Live reads through the Mac node's wRPC (`ws://127.0.0.1:28640`) and the log intake (Neon HTTP SQL, lines split server-side), never a restart.
@ -1525,6 +1551,679 @@ What is measured: one BLS12-381 aggregate signature over 16 summed G1 keys plus
Reading (the NEW finding, ledger C4). With the module off GHOSTDAG alone converges on the heavier chain and the losing side's records re-determine (F24 works when the chain moves). With the module on the overlay holds during the split (A, with 30% of the frozen table, locks nothing; B locks 7 and 8) and then fails at the heal in the shipped node: B's certificates for blocks off n0's chain are "kept pending until the chain decides (no lock at this index)", n0's chain never decides because GHOSTDAG keeps its heavier tip and nothing turns the certificate into a fork-choice constraint, and once n0's last lock (index 7, DAA 209) is one window old (DAA 329) the frozen table stops applying on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys are 100% of A's own window (B's post-cut blocks are red there) and n0 locks 10, 11, 12 alone; B's certificates for 10 and 11 then log CONFLICTING on n0 (n0 log, 17:27:04 to 17:29:54 BST). A finality fork from a 96-s honest partition, no attacker, table intact at the heal; the 150-s run and the v2 control end the same way. The spec's fork choice ("GHOSTDAG among tips through all certified checkpoints", 3.5) is therefore implemented only for certificates over blocks already on the node's chain. Fix named in the ledger entry: verify an off-chain certificate against the table at its own block and let it constrain fork choice (a certificate-driven reorg), then re-determine. Raw: `scratchpad fud-a/c4-results-*.md`, node logs `c4-on90-tmp/`, `c4-v2-control-tmp/`.
## 5 October 2026 (evening), the 9070 XT on the eGPU: why 17.9 MH/s, and what moved
PC 1 (ae432dc7, Windows 11, Ryzen 7 9800X3D with its gfx1036, RTX 5090 on CUDA), an AMD Radeon RX 9070 XT (gfx1201, RDNA 4) in a Sonnet Breakaway Box 850T5 over USB4, Adrenalin 26.9.2 (OpenCL driver string `3683.0 (PAL,LC)`, platform `OpenCL 2.1 AMD-APP (3683.0)`). Branch `opencl-rdna4`. the project lead: "the hashrate is low" (17.9 MH/s with one worker; two workers on the card earlier gave 8.9 and 9.4).
**Before, from PC 1's own app log** (`node tools/logs.mjs win-ae432dc7-20261005-181046`, the miner's STATUS line for the card `amd:1:gfx1201`, 2^21-nonce jobs): `hash=17.82 MH/s wall (17.83 MH/s inside jobs) ... idle=0.3%`. Wall equals inside, so the host loop (template fetch, job line, read-back, scan) costs nothing measurable; the dispatch itself is slow. The worker's `ready` line: `exchange 0` (local memory: AMD lists `cl_khr_subgroups` and no shuffle extension), `batch 4194304`, `dataset-log2 28` (1 GiB), device `[1] gfx1201` on the 3683.0 platform, `AMD wavefront width 32`. The same card was listed again as `[3] gfx1201` on the older platform `3652.0` (the 32.0.21042 driver's OpenCL registration is still present after the update): that is the two-worker run.
**Hypotheses, each with its number** (the measurement job `rdna4-bench-1`, 18:39:25 to 18:41:17 UTC, the card switched off in the app through `POST /api/cards` for key `amd:1:gfx1201` only, the 5090 untouched; worker exe sha256 `53c7e8c9…5403e10` built from this branch by `proto-cuda/nvrtc/build-windows.sh`; read back with `node tools/jobs.mjs rdna4-bench-1`):
| # | Hypothesis | Measured | Verdict |
|---|---|---|---|
| 1 | The dataset or program is re-sent over the eGPU link per job | Nothing is re-sent: the dataset (1 GiB) and cache (256 MiB) are built on the device once per pair (`info first pack ... cache 11 dataset 51 ms` on the Mac check); per 2^21-nonce job the old path sent 32 B up and read 16 MiB down; the serve A/B below puts a number on that read-back | Not the cause |
| 2 | Work-group, occupancy, wave width, the exchange | `clGetKernelSubGroupInfoKHR`: sub-group 32 for a 32-item work-group (wave32), private memory 0 (no spills), preferred multiple 32; `--group-warps 1, 2, 4, 8` = 18.024, 18.063, 18.070, 18.039 MH/s (`--batches 3`, 2^24, device event time); `--batch-log2 21` (the app's job size) = 18.108 | Not the cause: the shape does not move the number |
| 3 | The wrong AMD platform | The app's worker runs on `[1]`, the 3683.0 platform (ready line). The old platform's `[3]` gives 18.049 MH/s: the same. The duplicate listing is real and is the two-worker halving | Not the cause of 17.9; fixed anyway (below) |
| 4 | The card's own random-read rate | `--memprobe`: dependent random 4-byte loads over 1024 MiB top out at 2.42 to 2.68 G loads/s from 4,096 lanes up (table below); 128 loads per hash gives a ceiling of 18.9 to 20.9 MH/s; the hash runs at 18.0 to 18.1 | THE CAUSE: the hash is at 87 to 95% of what this card does for this access pattern |
**The memprobe on the 9070 XT** (`igneum-worker-opencl.exe --device 1 --memprobe`, device event time, best of 3, 256 dependent steps per lane; `chase` = one dependent random 4-byte load per step, `indep x8` = eight independent chains per lane):
| Buffer | Work-group | Lanes in flight | chase G loads/s | ns per dependent load | indep x8 G loads/s |
|---|---|---|---|---|---|
| 4 MiB (inside the 8 MB L2, approximate size) | 256 | 4,096 | 34.95 | 117 | |
| 4 MiB | 256 | 262,144 | 64.63 | 4,056 | 63.8 (262k lanes) |
| 64 MiB (the 64 MB Infinity Cache, approximate size) | 256 | 4,096 | 9.17 | 447 | |
| 64 MiB | 256 | 262,144 | 9.18 | 28,561 | 8.8 (262k lanes) |
| 1024 MiB (GDDR6) | 32 | 4,096 | 2.64 | 1,552 | |
| 1024 MiB | 32 | 65,536 | 2.60 | 25,181 | |
| 1024 MiB | 32 | 4,194,304 | 2.43 | 1,729,136 | |
| 1024 MiB | 256 | 4,096 | 2.64 | 1,552 | |
| 1024 MiB | 256 | 262,144 | 2.45 | 106,831 | 2.46 (262k lanes) |
| 1024 MiB | 256 | 4,194,304 | 2.42 | 1,732,023 | 2.42 (4M lanes) |
| ALU chain, 1,048,576 lanes x 4,096 steps | 256 | | 6,219 G int ops/s (5 ops per step counted, approximate) | | |
Reading: at the dataset size the card delivers about 2.5 G random 4-byte reads per second whatever the parallelism (4,096 lanes already saturate it; more lanes only queue, the ns column is Little's law on a fixed throughput). Eight independent loads per lane give the same 2.4 G/s, so it is not a latency-hiding problem in the kernel. Inside the Infinity Cache the same chain runs 3.7x faster and inside L2 26x faster, so the cap is the path to GDDR6 for random reads. The ALU chain says the shader clock is not parked (approximate: 6.2 T int ops/s is of the order of 64 CUs x 64 lanes x 2.46 GHz with quarter-rate multiplies).
**Against the other two cards** (same probe; the 5090 through NVIDIA's OpenCL `[4]` WHILE its CUDA worker was mining, so a lower bound; the Mac through Apple OpenCL, wall time, a Mac at high load, approximate):
| Card | 1024 MiB chase at 4,096 lanes | 1024 MiB chase ceiling | indep x8 ceiling | ceiling / 128 = hash ceiling | measured hash rate |
|---|---|---|---|---|---|
| RX 9070 XT, eGPU over USB4 | 2.64 G/s, 1,552 ns | 2.42 to 2.68 G/s | 2.42 G/s | 18.9 to 20.9 MH/s | 18.0 to 18.1 MH/s (bench), 17.8 (app) |
| RTX 5090, PCIe 5 x16, contended | 9.09 G/s, 451 ns | 16.4 to 18.0 G/s | 16.2 to 16.7 G/s | 128 to 141 MH/s | 127 MH/s (app, the project lead), 139.7 alone (M11) |
| Apple M5 Max, Apple OpenCL | 2.10 G/s, 1,949 ns | 3.41 to 3.49 G/s | 3.45 to 3.47 G/s | 26.6 to 27.3 MH/s | 27.9 Mhash/s (README, Apple OpenCL) |
Reading: on all three cards the hash runs within a few percent of 1/128 of the card's dependent random-read ceiling, which is what a 128-load program should do; the probe is a good model of the hash. The 5090 does 6.6x the random reads of the 9070 XT for 2.8x the rated bandwidth (1,792 against 640 GB/s, vendor figures): the rest is access granularity and DRAM behaviour on random 4-byte reads, which the kernel cannot change.
**Power, heat, fans and clocks, measured** (branch `opencl-rdna4-telemetry`; the project lead watched the 9070 XT at 90% usage with its fans barely turning and the app had no AMD reading, the MH/W line came from nvidia-smi only; a new helper `proto-opencl/gpu-telemetry.c` reads ADLX on Windows and the amdgpu sysfs on Linux. Job `tele-measure-1`, 20:27:45 to 20:29:41 UTC, both cards mining in the app, nothing touched: `igneum-gpu-telemetry -l 5` (sha256 `703cf69c…a9c69b`) and `nvidia-smi --query-gpu=index,name,power.draw,temperature.gpu,fan.speed,clocks.mem,clocks.gr,utilization.gpu -l 5` side by side, the app's `hash_now` every 5 s; `node tools/jobs.mjs tele-measure-1`):
| Card | Samples | Watts (mean, min to max) | Temperature | Fan | Memory clock | Shader clock | Busy | Hash (mean of 24) | MH/W, measured |
|---|---|---|---|---|---|---|---|---|---|
| RX 9070 XT, bus 98, ADLX `GPUPower` | 12 (the helper's buffered tail was lost at the kill; fixed, `fflush` per sample) | 198.9 (193 to 212) | 64 C | 657 rpm (ADLX gives rpm; no percent) | 2,505 MHz | 3,290 MHz | 100% | 17.73 MH/s | 0.089 |
| RTX 5090, nvidia-smi, 450 W cap | 24 | 307.6 (306.3 to 308.7) | 69 C | 44% | 13,801 MHz | 2,850 MHz | 94% | 122.30 MH/s | 0.398 |
| gfx1036 (integrated, idle) | 12 | 42.7 (32 to 56; the package, not the GPU alone) | 62 C | none | 2,800 MHz | 600 MHz | 0% | off | |
Reading: the 9070 XT draws 199 W of its 304 W board rating (vendor figure) at 100% busy with the shader clock at its top, so the die is waiting on memory, which is the ceiling finding again; the fans at 657 rpm and 64 C are the card's own curve at that load, not a fault. Per watt the 5090 is 4.5x the 9070 XT on this program class (0.398 against 0.089 MH/W). The earlier per-watt claim from the board rating (304 W) would have read 0.058 MH/W; the measured number is 1.5x that.
**Is it the eGPU link?** No. 2.42 G loads/s x 64 B lines = 155 GB/s of DRAM traffic, forty times what a USB4 PCIe tunnel carries (about 4 GB/s, approximate); the 1 GiB buffer sits in the card's own memory (the 4 and 64 MiB cases show the card's caches at work above it, and a buffer in host memory would run below 0.1 G/s). A PCIe slot would move the per-job read-back (16 MiB per 2^21-nonce job on the old path, now gone) and nothing else; the random-read ceiling is the card's. What a PCIe slot would give: the same 18 MH/s.
**What changed on `opencl-rdna4`** (`proto-opencl/host.c`, `app/igneum-app/src/detect.rs`):
| Change | Before | After |
|---|---|---|
| Duplicate platform | `--list` showed the card twice ([1] 3683.0 and [3] 3652.0); the app made two cards and ran two workers (8.9 + 9.4 MH/s) | the older platform's entry prints as ` dup [3] ... hidden, use [1]`, the default pick skips it, the app's parser (`parse_opencl_list`, 3 tests) never makes a card of it; `--device 3` still works for comparison. Verified on PC 1: `platforms: 2 device(s) hidden ...`, cards `amd:0:gfx1036` and `amd:1:gfx1201` only |
| Kernel report | work-group and local memory | plus preferred multiple, private memory (spills), sub-group size on every exchange path (`info kernel:` in serve mode) |
| Read-back per dispatch | 8 B per nonce (16 MiB per job) and a host scan of 2^21 words | a GPU select pass: the hits (index, hash) behind an atomic counter plus 34 sentinel words; 276 B per chunk plus 16 B per hit; found lines in nonce order; `--readback full` / `IGNEUM_READBACK=full` keeps the old path; a chunk with over 256 hits falls back to the full read |
| Transfer accounting | none | bytes up and down per chunk and the mean device time of kernel, select, read-back and scan in the stats line every 200 jobs and at quit |
| `--memprobe` | none | the tables above, no pack needed |
Correctness: `proto-opencl/test-generic.sh` on the Mac (Apple OpenCL) PASS on both paths: "15 sampled hashes (both packs, both sides of the 32-bit nonce boundary) equal igneum-pow hash-bound"; select path transfers `5 chunks, up 180 B, down 4452 B`, full path `up 160 B, down 1536 B` (the check's jobs are 32 to 64 nonces with every nonce a hit). The bench on the 9070 XT: cache check PASS, dataset self-test PASS, 6 of 6 vector warps PASS, batch fingerprint `3cc4fbf90fa6366c` at 2^24 for the devnet pack (the Apple OpenCL value in the README), at every `--group-warps`.
**The serve-mode A/B on the card** (job `rdna4-serve-4`, 19:11 UTC, card off in the app, worker exe sha256 `324a6d9b…2bfdfff`; 200 real `job` lines of 2,097,152 nonces each, the app's `--job-nonces`, against the emulator test pack `pack-a` (epoch `edc4fa84…`, self-test PASS, 96 of 96 vector lanes), target `0000100000000000` so that 408 hits fall in 200 jobs on both paths; `done` ms over jobs 11 to 200; `node tools/jobs.mjs rdna4-serve-4`):
| Read-back | Bytes down per job | Kernel (device, mean) | Select pass | Read-back (wall) | Host scan | Mean job | Inside-job rate |
|---|---|---|---|---|---|---|---|
| full (before) | 16,777,216 | 116.12 ms | 0 | 7.28 ms | 0.55 ms | 124.22 ms | 16.88 MH/s |
| select (after) | 309 | 116.00 ms | 0.039 ms | 0.78 ms | 0.00 ms | 117.38 ms | 17.87 MH/s |
| select (repeat) | 309 | 115.96 ms | 0.038 ms | 0.76 ms | 0.00 ms | 117.33 ms | 17.87 MH/s |
Reading: the kernel is the same 116.0 ms on both paths (18.08 MH/s pure kernel, the bench's number). The old path paid 7.8 ms per job for 16 MiB over the eGPU link (2.3 GB/s, the USB4 tunnel's rate; a PCIe slot would read it in about 1 ms, approximate) and the host scan. The select pass removes it: +5.9% per job on this link, nothing on the kernel. Both paths found the same 408 hits. The `--group-warps` and exchange levers were already shown flat above, so this is the whole host-side gain available on the 9070 XT.
**Probes with a fresh seed per repetition** (the first probe round replayed the same addresses on repeats, so its low-lane rows were cache hits; fixed in `probeLaunch`, job `rdna4-serve-4`): 1024 MiB chase at 256 lanes 276 ns per dependent load, at 1,024 lanes 422 ns, at 4,096 lanes 1,560 ns (2.63 G/s, the cap). Random 64-byte lines (four `uint4` loads per step) at 1024 MiB: 2.46 to 2.88 G lines/s = 158 to 184 GB/s in lines, the same count per second as the 4-byte chase: every random 4-byte read costs this card a 64-byte line fetch. Coalesced stream over the whole 1024 MiB: 635.2 GB/s against the vendor's 640 GB/s, so the memory clock is in its full state and the card is not parked. Inside the 64 MiB buffer the line probe reaches 8.3 to 14.0 G lines/s (533 to 894 GB/s in lines: the Infinity Cache, approximate).
**A second defect found on the way: the pack export race.** PC 1's app log since its 19:02 UTC restart (`node tools/logs.mjs win-ae432dc7-20261005-190232`): `worker error: error 0 pack packs\devnet: the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT` at 19:07:03, 19:07:19 and 19:08:07, so the 9070 XT was not mining at all in the app while this entry was written (my job `rdna4-serve-1` at 18:43 hit the same folder in the same state). Cause, from `app/igneum-app/src/engine.rs` `prepare_worker`: one thread per card, each running `igneum-miner export-pack` into the one folder `packs\devnet`; across an epoch change the two exports interleave and the folder keeps one epoch's `program.h` with the other's `seeds.txt` until the next export. Fix on this branch: a process-wide mutex around both export sites (`EXPORT_LOCK`); the second export rewrites the same pack. Not measured in the app yet: it ships with the branch.
**Answer to the project lead.** The 9070 XT does 2.5 G random 4-byte reads per second from its memory for this access pattern, and the hash needs 128 of them, so about 19 MH/s is this card's ceiling for the current program class, on any slot; it was running at 92% of that. The eGPU link cost 6% per job through the read-back, now removed (17.87 against 16.88 MH/s inside jobs standalone). The duplicate platform that halved it to 8.9 + 9.4 is folded away. The pack race that stopped it is serialised. Nothing else in the worker's control moves the number: the next step for this card is the program class itself (fewer, wider loads per hash would favour AMD's 64-byte lines), which is a consensus question, not a worker one.
## 5 October 2026 (night), Ember Tune: the two-knob efficiency tune, the fleet prior, and what PC 1 could measure tonight (miner-community-lead)
Branch `ember-tune` (54ff1bc), docs/plans/ember-tune.md. Every card tuned for MH per watt out of the box: the power limit and the core clock cap stepped on the live kernel (memory clock never touched), the point with the best MH per watt within 1% of the top rate kept and pinned, every result uploaded as a `TUNE {json}` record (a hash of the install id, no address) and folded per (card model, driver major, program class) into a prior the signed manifest carries back, so a new card of a known model starts there and confirms it in two steps.
**What was measured tonight (PC 1, machine ae432dc7, from its own uploads to the intake):**
| Fact | Where it was read | Consequence |
|---|---|---|
| The installed 0.3.9 app runs as `DESKTOP-KMCV30N\Admin` with `elevated=False` (account line, 19:02:33 UTC) | app log `win-ae432dc7-20261005-190232` | `nvidia-smi -pl` and `-lgc` need administrator rights; the one prompt is the Power control switch (3562f26), which the app never raises by itself |
| Two in-app sweep attempts aborted at 20:09 UTC: `the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)` | the same log | no stored sweep result from today exists; the 5090's two-knob tune is owed to the morning (one click on Power control, then it runs by itself within 2 minutes of steady mining) |
| The RX 9070 XT left PC 1's bus at about 20:40 UTC, was back at 21:09 and gone again at 21:22:59 UTC (the eGPU link, third drop today) | the telemetry agent and the PC 1 scheduler | the AMD path (ADLX, no prompt) is unit-tested on the helper's captured line shapes; its end-to-end run waits for the card |
**The pipeline, verified without a card:** 9 `ember` unit tests (plans, clamps, the choice rule, the five marks, a faulted step reverted inside a fake-clock run, the confirm verdicts, the baseline plan, the record and prior shapes, the vendor reasons), the AMD `tune` line and the 0.3.10 sample line parsed (`engine::amd_telemetry_tests`), the helper protocol (`sweep::tests`), 6 relay aggregation tests (five samples converge on 2,470 MHz at 100%; an outlier at 0.908 MH/W moves the median by nothing; baseline records make no prior; de-duplication; the manifest merge keeps lever 2's cards; the canonical round trip), 3 UI line tests. A test manifest was signed on this Mac with `packaging/ota/publish-manifest.sh --tuning` from fixture priors: `tuning.priors["NVIDIA_GeForce_RTX_5090|581|l128w16"]` = 2,470 MHz at 100%, 5 samples, beside the kernel-variant `cards` entry and `tuning.ember {enabled: true, min_samples: 5, rate_tolerance_pct: 1}`, signature verified by the signer, 21:25 UTC.
**Tier consequences** (docs/plans/ember-tune.md section 7): a 9-step full tune costs about 12 minutes once and 3 minutes a week per card, under 1% of the hour, the worker never stops; a rig tunes one card at a time and every card of a known model after the first takes the 3-minute confirm; a pool user gives up the same 1% of shares at most; Apple silicon and AMD on Linux measure only and the row says so.
**The PC 1 run, 22:30 UTC (job ember-tune-pc1-1, engine aeea3228..., PC 1 on 0.3.10):** the job published at 22:29:40Z, the installed app stopped its miners and started the second engine at 22:30:21Z, and at 22:31:06Z the installed app quit (its log: `quit: stopping the miners, then the node`, then `job ember-tune-pc1-1: aborted (the app is quitting)`), 46 s in, before any step. Nothing was set. Corrected the same night (C35), then named the next morning from the second engine's own log (collect ember-c35-collect-1, 06:59Z): the second engine, reporting 0.3.9 (the branch's Cargo version) under the manifest's `min_supported_version`, took the 0.3.10 update as urgent (the "urgent" rule beats the copied `auto_update = false`), downloaded it at 22:31:02Z and started `ota-apply.ps1` with the per-user installer at 22:31:05Z; the installer's PrepareToInstall sent `POST /api/quit` to the installed app, which logged `quit:` at 22:31:06Z. So the source was my own second engine's updater, through the installer, one second before: a second install of 0.3.10 over the 0.3.10 PC 1 had taken through the shipper's update-now at 21:40:41Z (release-0.3.10.md section 8), whose only effect was the quit and the hang. The first reading (the 0.3.11 rollout) was wrong in the cause and right in the class: an installer. What else is established: the engine's quit then HUNG for 24 minutes in the jobs runner's abort, waiting for EOF on the script's stdout pipe whose write end the second engine and its miners had inherited, and those miners (2 igneum-miner, 2 CUDA workers, 1 OpenCL worker) mined on, orphaned, until the relay lane killed them at about 23:00Z; the second engine also raised one administrator prompt at about 22:30:25Z (`apply_power_limits` at start counted `--sweep` as Power control), 41 s before the quit; PC 2's unexplained quit at 20:01:09Z came 20 s after a cancelled prompt of the same class, so the prompt is the common factor and the morning's test (one prompt raised beside the mining app on PC 2, the stamped quit line read). Fixed on the branch: b671c8b (quit sources, Power control alone decides, no cap at start under `--sweep`), 8ab9068 (no pipe into a second engine, its tree ended, the CI check), and the third close: a second engine never runs the updater (`IGNEUM_APP_NO_OTA=1`, implied by `--sweep`; the playbooks set it; the CI check demands it). What the run did record, the "before" snapshots with the miners stopped:
| Card | Read back at 22:30:20Z | Meaning |
|---|---|---|
| RTX 5090 (driver 617.14) | limit 450 W of 575 W default (min 400, max 600), draw 259.9 W idle-after-stop, core 2,850 MHz, `clocks.max.gr` 3,090 MHz, memory 14,001 MHz | the two-knob plan for this card is 5 power steps (575, 518, 460, 403, 400 W) and 4 clock steps (2,781, 2,472, 2,163, 1,854 MHz); it needs the one administrator prompt (Power control) |
| RX 9070 XT (bus 98, present again) | `tune 1 ... gmax 0 gmax_range -500 1000 plimit 0 plimit_range -30 10 factory 1 ok` | the helper's clock range is an OFFSET from stock in MHz, not a ceiling: a probe reading it as a 1,000 MHz maximum would have asked for `--set-gmax 900`, an overclock. Fixed at 054e041: an offset range closes the clock knob (until the stock clock is known) and the power ladder runs on the percent scale bounded by the range, so the 9070 XT's plan is 100, 90, 80, 70% (the -30 floor), 4 steps |
| Radeon(TM) Graphics (integrated) | `tune 0 ... gmax - ... factory 0 ok` | no manual tuning: measure only, and it is off by default anyway |
**Run 2, 6 October 2026, 07:21 to 07:56Z (job ember-tune-pc1-2, elevated on the project lead's word, engine 25113f52..., PC 1 on 0.3.11):** the project lead answered the one prompt; the installed app stopped its miners at 07:21:16Z; the second engine ran for the whole 35-minute budget at "waiting, 0.00 MH/s" and no step ran. Cause: the playbook wrote the engine's copy of settings.json with PowerShell 5.1's `Set-Content -Encoding utf8`, which adds a UTF-8 BOM; the engine's JSON parser refuses it, `Settings::load` fell back to defaults (no payout address, no cards), the engine logged `[error] no payout address` and never started a miner. Run 1's scratch log carried the same line the night before. Readbacks, idle both times: the 5090 at 90.6 W before and 69.9 W after (2,505 then 2,407 MHz core, 14,001 MHz memory, limit 450 W of 575), the 9070 XT at factory (`gmax 0`, `plimit 0`). Nothing set on either card. The installed app's runner released the miners-stopped hold by itself on the failed exit (`job finished; the miners restart` at 07:56:50Z, both miners up by 07:57:04Z, `mining` at 07:57:29Z): mining paused 36 min 13 s. Fix 8273494: the copy is written without a BOM, the address is read back and the job fails within seconds if it is empty (`RESULT TUNE scratch settings: address ..., cards N, first bytes ...`), and the CI check fails any playbook writing JSON with `Set-Content -Encoding utf8`. The re-run needs one more click on the prompt.
Consequence for the tiers: an AMD card is tuned on its power limit alone until its stock core clock is read (a 9070 XT at -30% is the floor the driver allows, 4 steps, 5 minutes); every NVIDIA card's two-knob plan waits on the user's one click on Power control; the re-run on PC 1 is held until the quit's source is named (the event-log collect) and follows the 0.3.11 rollout (the update clears the jobs folder, so the engine and the helper are fetched again), with the scheduler's slot.
## 5 October 2026 (night), read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, a written scratch; three cards (gate 1 experiment, cryptographer)
Branch `readwidth` (commits 019b014, b970dda, 4badcee, a9e002c, d0018cf and the entry commit); plan and recommendation in `docs/plans/read-width.md`. Nothing here changes consensus: every class sits behind `igneum-pow --class` and the default class is generator version 2 byte for byte (`igneum-pow/tests/packs.rs` passes on the four pinned packs after every commit). Question (the project lead, after "the 9070 XT on the eGPU" above): would wider reads keep the latency-bound random-access property while closing the vendor gap. Additions from the coordinator: a per-load width drawn from an era-fixed mix, and a written per-warp scratch (measurement only, no soundness claim).
**What a class does** (`igneum-pow/src/generator.rs` `LoadClass`, `verify::fold_words`, the three emitters): a load of W words reads the W-word-aligned address `(src AND MASK) AND NOT (W - 1)` and folds every word into `dst` (`x = dst ^ w0; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]`); W = 1 is the lottery hash exactly (`w4` = pack `bcc1248b10cc90f2`). A mix class draws W per load with one extra `below(100)` roll per instruction. A scratch class `scr<k>k<kb>` turns `k` of the 16 memory slots into read-modify-writes of a 16-byte slot of the lane's share of a `kb` KiB per-warp scratch (kernels run persistent warps, one per block or work-group; a slot reads as a seed-and-base fill until the unit writes it, behind a per-unit tag). Program ids carry the class. Dependent chain and 32-lane unit unchanged.
**Correctness**: 23 packs (`proto-cuda/packs-readwidth/`, Rust CPU reference vectors). Every pack passed its three vector units and the cache and dataset checks on Metal (M5 Max, `proto-metal/packbench`), Apple OpenCL (`--bench-pack`), the RTX 5090 (NVRTC, `igneum-worker-cuda --bench`) and, the 16 width and mix packs, the RX 9070 XT (`igneum-worker-opencl --bench-pack`); the 2^24 batch fingerprints agree across all four runtimes on every pack (for example w16 `e7c890445b47af60`, w64 `836e56e7d496e980`, mixB-2 `a18ac73098c76007`). The clang CUDA emulation (w16, w64, w64x4, mixA-0, mixB-0: 3 of 3 units standalone and 2 of 2 in batch at 2 warps per block) and the clang OpenCL emulation (the same five plus scr2k32 and scr8k128, sub-group 32 and, width packs, wave64 with sub-group shuffles) pass with equal fingerprints per configuration. Acceptance rule on the classes: 60 candidates per class, rejection 0 to 14 of 60 (w16 and w64 as v2; the mixes the same; the scratch classes' distinct-address bound now covers dataset loads only, since a 64-slot lane scratch repeats slots by design). CPU verifier (M5 Max, one core, avg of 50 units, `igneum-pow bench --class`): v2 0.604 ms, w16 0.610, w64 0.630, w64x4 0.160, mix50-35-15 0.620, mix25-50-25 0.614, scr0k32 0.600 (1.004 on a loaded re-run), scr2k32 0.657, scr4k32 0.458, scr8k32 0.317, scr2k128 0.535, scr4k128 0.458, scr8k128 0.311; per hash divide by 32. The wide reads cost the verifier nothing (a lane's words lie in one item); scratch ops replace item derivations and make it cheaper.
**Probes** (`--memprobe`, dependent random reads at 1024 MiB, G reads/s, best over lanes in flight; 4 B = the hash's pattern; the 5090 and 9070 XT with the card off in the app, the Mac through Apple OpenCL under a load average of 5 to 10):
| Card | 4 B chase | 16 B | 64 B | 64 B as GB/s | coalesced stream GB/s (rated) | integer chain |
|---|---|---|---|---|---|---|
| RTX 5090 (PC 2, CUDA) | 17.5 to 18.2 | 18.0 to 19.9 | 9.1 to 15.7 (9.1 at 4 M lanes) | 584 | 1,579 (1,792) | 39.0 T op/s |
| RX 9070 XT (PC 1, eGPU, OpenCL) | 2.42 to 2.66 | 2.43 to 2.73 | 2.47 to 2.87 | 158 | 636 (640) | 6.2 T op/s |
| Apple M5 Max (Apple OpenCL, approximate) | 3.50 | 3.51 | 3.51 | 225 | 522 | |
Reading: on the 9070 XT and the M5 Max a 64-byte dependent read costs exactly what a 4-byte one costs (the line is fetched either way); on the 5090 a 64-byte read costs about two 4-byte reads (two 32-byte sectors) and the 64 B chase at full occupancy sits at 584 GB/s, a third of the stream.
**Hash rates** (5 timed dispatches of 2^24 nonces after a warm-up; Metal and the 9070 XT by device time, the 5090 by wall time around the stream sync; the PC cards switched off in the app for the run and restored, PC 1's 5090 and the integrated chip kept mining; the Mac under other agents' builds, load 4 to 9, so its absolute numbers carry that; the share = measured / (the card's probe ceiling at the class's widths / loads per hash)):
| Class | dataset B/hash | RTX 5090 MH/s (share) | RX 9070 XT MH/s (share) | M5 Max Metal MH/s (share) | 5090 / 9070 |
|---|---|---|---|---|---|
| v2 (w4, the lottery hash) | 512 | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x |
| w16 | 2,048 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x |
| w64 | 8,192 | 71.9 (0.58) | 17.59 (0.78) | 28.27 (1.03) | 4.1x |
| w64x4 (32 loads) | 2,048 | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x |
| mix50-35-15, 6 programs: min / median / max (spread of median) | 1,664 to 3,680 | 99.5 / 114.2 / 121.0 (18.8%) | 17.45 / 18.76 / 18.83 (7.4%) | 25.36 / 27.26 / 28.43 (11.3%) | 6.1x |
| mix25-50-25, 6 programs | 2,240 to 5,024 | 95.9 / 107.3 / 119.8 (22.3%) | 17.84 / 18.45 / 18.85 (5.5%) | 23.21 / 24.68 / 25.21 (8.1%) | 5.8x |
Scratch (variant 5; N persistent warps; 5090: 2,048 warps launched against a resident capacity of 4,080 = 24 blocks/SM x 1 warp/block x 170 SMs at `--block-warps 1`, the occupancy query unchanged by the allocation (24 before and after); Metal: 2,048 to 16,384 warps swept, best shown; arena = N x per-warp size; the whole working set = 1 GiB dataset + 256 MiB cache + 128 MiB output + arena, under 2 GB on every row):
| Class (k of 16 slots, KiB per warp) | scratch ops/hash | dataset B/hash | RTX 5090 MH/s (vs scr0, share) | M5 Max Metal MH/s (vs scr0) | RX 9070 XT MH/s | 5090 arena / working set |
|---|---|---|---|---|---|---|
| scr0k32 (control, persistent loop, no RMW) | 0 | 512 | 139.1 (0, 0.98) | 28.25 (0) | 17.88 (control, 0.86) | 64 MiB / 1.4 GiB |
| scr2k32 (12.5%) | 16 | 448 | 114.4 (-18%, 0.80) | 26.14 (-7%) | 14.65 (-18%) | 64 MiB / 1.4 GiB |
| scr4k32 (25%) | 32 | 384 | 109.8 (-21%, 0.76) | 31.74 (+12%) | 14.00 (-22%) | 64 MiB / 1.4 GiB |
| scr8k32 (50%) | 64 | 256 | 122.1 (-12%, 0.82) | 49.08 (+74%) | 14.17 (-21%) | 64 MiB / 1.4 GiB |
| scr2k128 (12.5%) | 16 | 448 | 110.1 (-21%, 0.77) | 26.24 (-7%) | 14.07 (-21%) | 256 MiB / 1.6 GiB |
| scr4k128 (25%) | 32 | 384 | 98.0 (-30%, 0.68) | 28.08 (-1%) | 13.14 (-27%) | 256 MiB / 1.6 GiB |
| scr8k128 (50%) | 64 | 256 | 72.8 (-48%, 0.49) | 35.44 (+25%) | 12.03 (-33%) | 256 MiB / 1.6 GiB |
The 9070 XT rows are 2,048 persistent warps (4,096 within 1 percent), arena 64 MiB at 32 KiB and 256 MiB at 128 KiB, working set 1.4 and 1.6 GiB; its control (17.88, the persistent loop) equals its v2 rate (18.15) within 2 percent, and every RMW share costs it 18 to 33 percent: on AMD a scratch op is a dependent 16-byte read plus a write into a region the 64 MB Infinity Cache does not hold for 2,048 warps, so it is memory work there as on the 5090, not the cached op it is on Apple. Apple OpenCL on the same scratch packs (wall time, `--bench-pack --warps 2048`): scr0k32 27.85, scr2k32 28.58, scr4k32 32.43, scr8k32 47.93, scr2k128 25.67, scr4k128 27.24, scr8k128 32.93 MH/s, the Metal shape within 4 percent, fingerprints equal. Bytes moved per scratch op: 16 read + 16 written (the tag word included); per hash at 50 percent, 1,024 read + 1,024 written beside 256 of dataset reads. The 5090 at 4,096 launched warps (above its 4,080 resident) lost 2 to 26 percent (scr8k32 90.0 MH/s), so the rows above are the in-capacity launch.
**Readings.** (1) Same count, wider: the vendor gap does not move at 16 B (7.8x) because on the 9070 XT a 4-byte read already costs a 64-byte line and on the 5090 a 16-byte read costs one 32-byte sector, the same as 4 bytes: the memory systems do identical work, only the fold's input grows. At 64 B the gap closes to 4.1x, entirely by the 5090 losing half its rate (its share falls to 0.58 and its DRAM traffic reaches 589 GB/s, 37 percent of the stream: bandwidth, not latency, bounds it), while the 9070 XT and the M5 Max do not move. (2) Fewer, wider (w64x4): 3.7x, but every card runs 4x faster because the dependent chain is 32 loads long instead of 128; the 5090 sits at a 0.56 share (bandwidth), so a chip with more bandwidth per dollar than a GPU gains, which is the Ethash shape the design avoids. (3) The mix: the hour-to-hour spread is 7 to 22 percent of the median per card (the 5090 the widest, because its 64-byte loads are the expensive ones and their count per program runs 2 to 8 of 16); the programs with many 64-byte loads (mixA-3, mixA-5, mixB-2) are the slow hours on the 5090 and the fast ones nowhere. (4) The scratch: on the 5090 every RMW share costs 12 to 48 percent against the persistent control, the 32 KiB arena less than the 128 KiB one (the smaller arena, 64 MiB over 2,048 warps, sits inside the 96 MB L2); on the M5 Max the 32 KiB rows are FASTER than the control (+12 and +74 percent at 25 and 50 percent), because the arena (128 MiB over 4,096 warps) lives in the chip's caches and a scratch op is cheaper than a dataset read, so replacing dataset loads raises the rate: the scratch at these sizes is not memory work on Apple and is partly cached on NVIDIA. The chip row for these variants comes from the ca2-soundness branch; what this entry gives is the GPU cost and the share. (5) Latency-bound shares: v2 0.87 to 1.01 on the three cards, w16 0.84 to 1.03, w64 0.58 (5090) and 0.78 (9070 XT); the Mac's shares above 1 are an Apple OpenCL probe under load against a Metal rate.
Jobs: `run-readwidth-5090-20261005` and `run-readwidth-9070-20261005` (probes; the packs refused for their string seeds, fixed in a9e002c), `run-readwidth-5090-20261005c`, `run-readwidth-9070-20261005c` (benches), `run-readwidth-9070-scratch-20261005d` (the scratch packs after the `__local` fix d0018cf, AMD's compiler requires the exchange buffer at the kernel's outermost scope); read back with `node tools/jobs.mjs <id> --all`. Mac commands and logs: `docs/plans/read-width.md` section 3. The worker exes for the jobs: `proto-cuda/nvrtc/build-windows.sh` on this branch (mingw), sha256 of the CUDA one `6f46336f...defe1`.
## 5 October 2026 (night), the hot table on the M5 Max: a second table sized to GPU cache beside the 1 GiB dataset (Counter ASIC 2.0 layer 5)
Branch `ca2-cache` (on readwidth 1ea7a52), `docs/plans/hot-table.md`. Apple M5 Max, measure lock held, the Mac's load average 14 to 27 throughout (other agents' CPU work; the lock serialises builds and measurements, not every process), so the ratios inside one session are the result and the absolute rates are not quiet numbers. Packs `proto-cuda/packs-ca2-hot/hot{32,64,96}k4`, `hot64k2`, `hot64k8` from `igneum-pow export --seed igneum-genesis --day 2026-10-03 --class hot<S>k<k>` (the version 2 genesis program with k of its 16 loads redirected to an S MiB table H keyed by `seed_words("igneum-hot/" || seed bytes)`, read at `H[mulhi(src, words)]`).
**Probe** (`proto-opencl/igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S`, Apple OpenCL, wall time, best of 3, 256 dependent steps per lane, work-group 256; the ceiling row is 4,194,304 lanes):
| MiB | chase at 4,096 lanes | ns per dependent load | chase ceiling, G loads/s | indep x8 ceiling | stream |
|---|---|---|---|---|---|
| 32 | 3.51 G/s | 1,168 | 21.7 | 21.8 | 138.7 GB/s |
| 64 | 3.63 | 1,129 | 12.8 | 13.0 | 199.1 |
| 96 | 3.30 | 1,242 | 12.3 | 12.7 | 242.6 |
| 1024 | 2.22 | 1,844 | 3.50 | 3.50 | 521.5 |
**Hash rate and bit-exactness** (Metal `proto-metal/packbench --pack <dir> --batches 5 --batch-log2 24`, GPU time; Apple OpenCL `--bench-pack --pack <dir> --batches 5 --batch-log2 24`, wall; both fill H on the device from the pack's `igneum_hot_fill` and check it; fingerprint = FNV-1a 64 over 2^24 outputs at base 0):
| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table head, last line, FNV | g against v2 (Metal) | probe-predicted g | ideal g |
|---|---|---|---|---|---|---|---|---|
| igneum-genesis-mh (v2) | 27.68 | 27.61 | 25f96e7dce90bd4e | 96/96 both | none | 1 | 1 | 1 |
| hot32k4 | 33.90 | 33.92 | d2e6cf3b61d0b9fe | 96/96 both | PASS both | 1.22 | 1.27 | 1.33 |
| hot64k4 | 30.93 | 30.89 | e4c5263ac650cc0d | 96/96 both | PASS both | 1.12 | 1.22 | 1.33 |
| hot96k4 | 29.06 | 28.97 | 5d63439b6e394521 | 96/96 both | PASS both | 1.05 | 1.22 | 1.33 |
| hot64k2 | 27.67 | 27.27 | 352633bdbbb0d2b6 | 96/96 both | PASS both | 1.00 | 1.10 | 1.14 |
| hot64k8 | 47.42 | 46.73 | da54630d7dfaaf85 | 96/96 both | PASS both | 1.71 | 1.57 | 2.0 |
Hot table fill, Metal GPU time: 0.07 ms (32 MiB), 0.15 (64), 0.22 (96). Hot table FNV-1a 64 of the genesis epoch: c1767ba3ef02719f (32 MiB), 77ca4b9527104530 (64), 79bcf436c4e5bc47 (96); cache 48c4f5bf24166b2e unchanged.
**CPU verifier** (`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class <c> --warps 50`, one core, release):
| Class | hot fill, one core | items per warp | ms per warp |
|---|---|---|---|
| v2 | none | 4,096 | 0.626 |
| hot32k4 | 24.0 ms | 3,072 | 0.489 |
| hot64k4 | 46.4 ms | 3,072 | 0.504 |
| hot96k4 | 73.0 ms | 3,072 | 0.488 |
| hot64k2 | 45.5 ms | 3,584 | 0.560 |
| hot64k8 | 47.7 ms | 2,048 | 0.344 |
Reading: bit-exact across Metal, Apple OpenCL and the Rust reference on every hot pack, hot table included. On this card the 32 MiB table delivers 92% of the probe's predicted gain with the dataset streaming beside it, 64 MiB about half, 96 MiB a quarter; k = 8 at 64 MiB gives 1.71x against an ideal 2.0x. The verifier gets cheaper with k (a hot load is one table read, a dataset load is an item derivation) and pays 24 to 73 ms per epoch for the fill. Chip model with these g in the plan, section 6.4. The RTX 5090 and RX 9070 XT rows are a prepared PC job (`relay/playbooks/ca2-hot-{5090,9070}-bench.ps1`, zip `~/Desktop/igneum-ca2-hot.zip`), not run.
Crate: `cargo test --release` 52 pass (39 unit, 13 pack tests: the four pinned v2 packs byte-identical, the five hot packs pinned with their load-form count: exactly 16 - k masked dataset loads and k hot loads per hash kernel).
**Addendum, the added form** (coordinator's form of 5 October 2026: 16 + k load slots, the k hot ones drawn among them, the 16 dataset loads and the 4,096-item verifier bound unchanged; packs `hot32k4a`, `hot64k4a`, `hot96k4a`; second Mac session 21:03 to 21:19 UTC, load average 7 to 14; same harnesses and commands, branch `ca2-cache` on ca2-v3 464d6e1, the hosts rebuilt on the merged packfile.h):
| Pack | Metal Mhash/s | Apple OpenCL Mhash/s | fingerprint (equal on both) | vectors | hot table | g against v2 (Metal, v2 27.63 in this session) | probe-predicted g | CPU verify ms/warp (v2 0.602) | hot fill, one core |
|---|---|---|---|---|---|---|---|---|---|
| hot32k4a | 25.76 | 25.72 | 8a3414735db4523c | 96/96 both | PASS both | 0.93 | 0.96 | 0.631 | 21.7 ms |
| hot64k4a | 23.92 | 23.87 | 45668f34105f6307 | 96/96 both | PASS both | 0.87 | 0.94 | 0.609 | 43.3 ms |
| hot96k4a | 22.92 | 22.88 | af763997dfee4c82 | 96/96 both | PASS both | 0.83 | 0.93 | 0.614 | 64.9 ms |
Reading: the added form costs this card 7, 13 and 17% of its rate at 32, 64 and 96 MiB for four extra loads per iteration, more than the probe predicts as the table grows; the verifier is unchanged (4,096 items, plus 32 table reads) and pays the fill per epoch. Chip arithmetic in the plan, section 6.4. All eight packs load and self-test through the rebuilt OpenCL host (the Windows exe's host.c) on the Mac.
**Addendum, the PCs** (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7, app 0.3.9 before and after; fetch `fetch-ca2-hot-20261005` (zip sha256 bd49faa1c9d48024f49c615481faff5c68a4c09f0889dbaf009c208674d67b3f), jobs `run-ca2-hot-5090-20261005` (126 s) and `run-ca2-hot-9070-20261005` (247 s), both exit 0, the card under test switched off in the app through `api/cards` and restored; workers `igneum-worker-cuda.exe` sha256 956c4ab34f42cbcd1d2c1c6fb1a58fd9b3a8c70166df771cafcd0296ca6a27d4 and `igneum-worker-opencl.exe` sha256 32d3d34390aad70485c3524424c354223387137d383b5c5daf01f40073c12703, built from ca2-cache 196db96 on ca2-v3's merged packfile.h d2cd6e1; read back with `node tools/jobs.mjs <id> --all`):
Probe (`--memprobe --probe-mib S`, dependent 4 B chase ceiling at 4,194,304 lanes, G loads/s; ns per dependent load at 4,096 lanes in brackets):
| Card | 32 MiB | 64 | 96 | 1024 | stream at 1024 MiB |
|---|---|---|---|---|---|
| RTX 5090 (CUDA, wall) | 112.6 (320) | 112.6 (340) | 112.6 (336) | 17.6 (610) | 1,563 GB/s |
| RX 9070 XT (OpenCL, event) | 9.88 (396) | 9.47 (457) | 8.18 (454) | 2.43 (1,579) | 633 GB/s |
Rates (5 dispatches of 2^24 after a warm-up; 5090 `--bench --block-warps 1`, 9070 XT `--bench-pack --device 1` work-group 256; every row check=PASS with the Mac's fingerprint; v2 references from the readwidth entry, same night, same workers: 136.1 and 18.15 MH/s):
| Pack | 5090 MH/s | g | 9070 XT MH/s | g | ideal g |
|---|---|---|---|---|---|
| hot32k4 | 146.6 | 1.08 | 19.79 | 1.09 | 1.33 |
| hot64k4 | 140.8 | 1.03 | 18.73 | 1.03 | 1.33 |
| hot96k4 | 138.5 | 1.02 | 18.33 | 1.01 | 1.33 |
| hot64k2 | 137.5 | 1.01 | 18.17 | 1.00 | 1.14 |
| hot64k8 | 163.6 | 1.20 | 22.32 | 1.23 | 2.0 |
| hot32k4a | 118.7 | 0.87 | 15.27 | 0.84 | 1 |
| hot64k4a | 115.4 | 0.85 | 14.62 | 0.81 | 1 |
| hot96k4a | 114.4 | 0.84 | 14.56 | 0.80 | 1 |
Reading: the probe promises a full hit rate on the 5090 (every S inside the 96 MiB L2 at one ceiling, 6.4x DRAM) and the hash gets 2 to 8% at k = 4 and 20% at k = 8; the 9070 XT the same shape. The dataset's random lines evict the table from the shared cache on every card. The added form costs 13 to 20% of the rate. Recommendation in `docs/plans/hot-table.md` section 6.4: do not adopt layer 5 in either form on these measurements.
## 5 October 2026 (night), mixer x4 and the cache growth rule: the class v3 dataset construction, with the x8 candidate (Counter ASIC 2.0; branch ca2-mixer on ca2-v3 6c75dad; cryptographer's lane)
Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0. Write-up `docs/plans/mixer-x4.md`; chip model `docs/analysis/chip-model-v3.md`; code `igneum-pow` (LoadClass mixer_mult and growth, memhard::Shape, the schedule, the three emitters), packs `proto-cuda/packs-ca2-mixer/`, tests `igneum-pow/tests/mixer.rs` and `tests/packs.rs`. Commits 0fc0ad1, 66eeba3, e4c04a7, 7ce8d1e, 504cae4, fe4e193 and this entry's.
What changed. Under program class v3 (`V3_CLASS = LoadClass::MX4`) every mixer application of the item derivation is `m = 4` applications with round keys `(r m + j + 1) x 0x9E3779B9`, the 8 dependent cache reads per item unchanged; the cache doubles when the dataset doubles (`growth_doublings(d) = floor(log2(1 + d / 1460))`: 2^26 words to day 1,459, 2^27 from day 1,460, 2^28 from day 4,380). Version 2 is byte-identical: fresh exports of igneum-genesis-mh and igneum-devnet-v4-epoch0 `diff -r` IDENTICAL against the checked-in packs, and the crate tests regenerate every pinned file. A v3 program of a seed is the v2 program of that seed instruction for instruction (v2 loads take no width roll); only the dataset words and the hashes change.
Bit-exactness, `with-lock.sh run`, 22:05 and 21:45 UTC: the two pinned v3 packs (mx4-genesis, mx4-devnet-epoch0: dataset words 0..15 `61ff2180 0d4c7e6c ...` and `afe80d67 b9fbd029 ...`, word MASK `5020180e` and `e6a99c7a`, unit at base 0 lane 0 `63acd2d273f475ba` and `212c6442b51e87ae`) and the two x8 candidate packs on Metal (`packbench`, built from this branch) and Apple OpenCL (`igneum-bench-cl --bench-pack`): 3/3 standalone and 3/3 in batch, 96 of 96 lanes, cache FNV-1a 64 unchanged from v2 (48c4f5bf24166b2e, 448274a57f508cbc), dataset head, word MASK and 64 samples PASS, one 2^24 fingerprint per pack across both harnesses (mx4 6f48d5a2aa0dbe5f and 73caaebb28e808fe; mx8 7c28cfb06c5c65a9 and bbb183f72692f840); hash rate the v2 rate (27.5 to 27.7 MH/s GPU time, the hash kernel is unchanged). Fuzz: 200 class v3 programs (4 units each across the 32-bit range, one in the top 256 nonces) interpreted twice on the CPU, 800 of 800; the same 200 packs on Metal 200 of 200 (`--batch-log2 9 --batch-base 4294967040`, the wrapping unit inside the window), every tenth on Apple OpenCL 20 of 20; x8: 50 of 50 on Metal, 5 of 5 on OpenCL. Stats (8,192 outputs per seed, two seeds): v3 avalanche 49.97 to 49.99 percent, worst bit z 1.92 to 3.09, 0 duplicates (v2 beside it 49.87 to 49.98, z 2.25 to 2.30). Edges: items 0, 1, 2^28 - 1, 2^32 - 1 by hand at m = 1, 2, 4, 8; words 0, 15, 16, 17, MASK - 1, MASK through the fetch path. Determinism: two epochs, every vector and file equal and equal to the pinned pack. The scratch soundness tests of ca2-soundness (cherry-pick 0d8f745) 7 of 7 on this tree. Crate: 44 lib + 12 packs + 4 mixer + 7 scratch tests pass. A first Metal fuzz run reported 200 of 200 FAIL on an empty RESULT line (a packbench built before the `--batch-base` cherry-pick); it was read as a failure, the harness rebuilt, the run repeated.
Timings, `with-lock.sh measure`, one session 21:40:12 to 21:40:23 UTC, one core, two rounds; the box carried a load average of 5.6 (one minute) and 26 (fifteen minutes) from unlocked processes, so the absolute figures are about 2.2x the quiet readwidth night's 0.604 ms v2 row and the ratios are the measurement:
| Construction | Verifier ms per 32-lane unit, avg of 50 (two rounds) | Worst cold unit | Against v2 | 256 MiB fill, one core | Metal 1 GiB build, GPU ms |
|---|---|---|---|---|---|
| v2 | 1.361 / 1.310 | 1.579 | 1 | 172 to 173 ms | 29.7 (first touch) / 21.0 |
| x4 (class v3) | 1.956 / 1.923 | 2.043 | 1.45x | 172 to 175 ms | 20.9 / 21.0 |
| x8 (candidate) | 2.785 / 2.790 | 2.942 | 2.09x | 172 ms | 21.9 / 21.9 |
Reading: the mixer multiplies the verifier's ALU part only (the 8 dependent misses per item are unchanged), hence 1.45x and 2.1x and not 4x and 8x; the Mac's GPU build is latency-bound and does not move with the mixer, so the "under 1 s on every discrete card" half of the x8 rule is the PC job (five packs, `relay/playbooks/mixer-x4-pc1-bench.ps1`, waiting for the go). Verification throughput (C19): a quiet 2026 core serves about 1,100 shares per second at x4 and 800 at x8 (1,660 at v2, re-cutting spec 09's 2,270), a 22,000-member pool at one share per 10 s needs 2 cores at x4 and 3 at x8, IBD over 108,000 headers is 1.6 min at x4 and 2.3 at x8 on that core; the 10 ms gate keeps 8.0 ms (x4) and 7.1 ms (x8) of margin on the loaded core, 6 to 7 ms on a 2019-class laptop core (approximate, unmeasured, O-1.14).
Chip model (`docs/analysis/chip-model-v3.md`): the on-die-cache recompute chip at 50 T op/s against the 5090's measured 136.1 MH/s: v2 334 MH/s, 2.45x bare, 7.4x with the 3x fixed-function factor; x4 83.5 MH/s, 0.61x bare, 1.84x with the factor, 1.53x with the 128 mm^2 N5 mirror deducted at equal silicon; x8 41.7 MH/s, 0.31x, 0.92x, 0.76x. The claim at x4 is "under 2x" with the margin thin on the equal-budget convention (a 3.3x factor or a 10 percent larger budget reads 2.0x); the hot table in the added form would have raised it to 2.1x to 2.2x at the 5090's g (kept as measured, not adopted). Nothing here is a measurement of a chip.
**Addendum, 22:15 UTC: the verifier regression, the PC 1 build rows, and x8 into v3.** The era agent measured the same v2 input with readwidth's binary (0.604 ms) and ca2-v3 HEAD's (1.33) in one minute; bisected under the measure lock to this branch's 0fc0ad1 (seam 6c75dad 0.610, 0fc0ad1 1.332; the "loaded box" reading above was wrong by that factor, the load was real but the 2x was the code). Cause: the item loop (`derive_items`) inlined into `MemhardCpu::fetch`; the mask hoisted, the mask constant, and the constant-mask loop inlined all stayed at 1.33, the same loop `#[inline(never)]` read 0.60 to 0.62. Fix: `derive_items_mask`, out of line, one instance per cache size with the line mask a constant. Measured the era agent's way (readwidth's binary beside the fixed one, same input, same minute, 22:07 UTC): v2 0.607 / 0.610 against 0.609 / 0.611; on the fixed binary x4 1.238 / 1.237 (2.0x), x8 2.077 / 2.058 (3.4x), worst cold 2.15 ms; the increments (+0.63, +1.46 ms per unit) equal the slow binary's. Lesson, the class: an inlined item loop costs 2.2x and nothing in the suite sees it; a verifier benchmark with a pinned bound in the crate's CI is filed for the next cut, and until then every change to the item loop is measured against the previous binary on the same input in the same minute. PC 1 (job run-mixer-x4-pc1-20261005, 22:00 to 22:04 UTC, the worker's `cache ... dataset ... ms` wall line): RTX 5090 dataset 23 to 25 ms at v2, x4 and x8; RX 9070 XT (gfx1201) 72 to 77 ms at all three; every fingerprint equal to the Mac's; rates the v2 rate (136.5 to 137.4 and 18.0 to 18.2 MH/s). Decision under the delegated rule (coordinator, 22:05 UTC): x8 enters class v3 (`V3_CLASS = MX8`); pinned packs mx8-genesis (7c28cfb06c5c65a9) and mx8-devnet-epoch0 through the chain path with the era inside (90f794dd556f7a3b, Metal and Apple OpenCL, 22:12 UTC); the x4 packs kept as the candidate's record. Chip headline at x8: 41.7 MH/s, 0.31x bare, 0.92x with the 3x factor, 0.76x at equal silicon (`docs/analysis/chip-model-v3.md`).
## 5 October 2026 (evening), EVM transaction relay: three nodes in a chain, every transaction sent to one end included by the other two miners (execution and networking engineer)
Until this change the node did not relay EVM transactions to its peers, so a transaction sent to one node was only ever included by that node's own templates (this file, "5 October 2026 (afternoon), live devnet: real transactions": 3,794 transfers, all in the Mac's blocks; execution-layer ledger item 9). Fork branch `tx-gossip` (worktree `vendor/igneum-node-txgossip`, from release-0.3.6 a24ab01a, commit e242acd0), main repo branch `tx-gossip`. Design in `docs/design/execution-layer.md` 1.4 "Relay"; the hand-out cooldown of its 10.2 table is gone with it (row "Mempool hold").
What was built. Three p2p messages after Kaspa's own transaction relay (`protocol/flows/src/v10/txrelay/flow.rs`): an inventory of admitted hashes, a request for the unknown ones, one answer with the raw bytes (`protocol/p2p/proto/p2p.proto`, payload numbers 72 to 74). Two flows per peer (`protocol/flows/src/v10/evmrelay.rs`), a pump that announces the mempool's admitted hashes every 250 ms, a sink trait the execution layer implements (`kaspa_consensus_core::evm::EvmTxSink`, `igneum/exec/src/service.rs EvmTxRelaySink`), and the mempool's side: every admitted hash queued for gossip, executed, evicted and invalid hashes remembered (65,536) so a second announcement is not requested, a 50,000-transaction cap, and the hold on block-added in place of the 4-second cooldown (the executor subscribes to consensus `BlockAdded`; a transaction leaves the templates when any DAG block carries it and comes back if a chain block skipped it). Limits per peer in the table.
| Limit | Value | Over it |
|---|---|---|
| Hashes announced to us, or requested from us | 2,000 per second, burst 8,192 | the surplus of the message is dropped (the sender paid as much as we did) |
| Hashes per inventory or request message | 4,096 | disconnect |
| Bytes per answer / per transaction | 4 MiB / 128 KiB | disconnect |
| Transaction failing a state-free rule (malformed, signature, chain id, type 3 or 4) | | disconnect, hash remembered |
| State-dependent refusal (nonce more than 16 ahead, fee cap under the base fee, funds, 64 queued per sender, pool full) | | dropped quietly, hash not remembered |
Protocol version. 13 to 14. An Igneum node drops a connection on a payload it cannot decode (`protocol/p2p/src/core/router.rs route_to_flow`: prost leaves the oneof empty, the router returns "empty payload", the connection closes), so the three messages go only to peers that advertised 14 or later, exactly as the finality (12) and proof-record (13) messages did. A 14 node registers the 13 flows for a 13 peer and never announces to it. The consensus params digest does not cover the protocol version: a scratch node on the devnet profile from the shipped 0.3.6 binary (`target-036`) and from this build printed the same digest, `9409dedac4bf9f0f20a54fb169b52a75a2903364909fe9ffd6fc5cdcd9d95d38`, so a 14 node and a 13 node still peer. Rollout: during the mixed fleet a transaction reaches the 14 nodes connected to the node it was sent to, and whatever a 13 node mines carries only what its own RPC received, as today; the relay is complete when the last miner is on 14. No fresh chain, no activation height.
Unit tests (PC 2, job build-20261005-173606, `igneum-exec` 15 of 15 in 0.01 s, `kaspa-p2p-flows` 33 of 33 in 0.19 s, 24 s for both): the pool queues an admitted hash for gossip once and answers "known" for the duplicate; wrong chain id, a signature above the curve order and truncated bytes are refused and remembered by hash, a nonce beyond the gap and a fee cap under the base fee are refused and not remembered; a transaction stays in every template until a block carries it, is held then, comes back when a chain block skips it and leaves (hash remembered) when one executes it; the wire messages round-trip through prost and the router's payload type, hash lists of the wrong length or over 4,096 are refused, the per-peer bucket grants the burst then the rate. The first PC 2 run of the suites (build-20261005-173013) failed on the signature case: a flipped low bit of `s` recovers a different signer (a funds refusal), not a fault; the test now sets `s` above the curve order. Found on the way: the `kaspa-p2p-flows` test target had not compiled since M20 added the epoch-seed headers to the pruning proof messages (`ibd/proof.rs` tests), fixed in the same commit.
The 3-node run (`tools/txgen/relay-net.mjs`, new; this Mac, load 7 to 8 at the end of the run after the other agents' harnesses finished, every number a count or an inclusion latency, not a timing of the node). Fast-time profile (`infra/fast-time/override-60x.json`, proof of work skipped), ports 29700+, data `/tmp/igneum-txrelay`. Chain A - B - C: B dials A and C (a harness node that dials accepts no inbound, and `--connect` takes one address per flag; both found by the first two runs, which are not numbers). A mines nothing. One vmine on B and one on C at 0.5 blocks/s each, paid to throwaway keys made for the run; B's rewards funded 16 generator wallets (2 IGN each) 33 s after start. The generator (`tools/txgen/run.mjs`) sent to A's EVM RPC only, 2 transfers a second for 120 s, so every inclusion is by a block B or C built from a pool the relay fed; C is two hops from A. Result files `docs/benchmarks/evm-relay-2026-10-05/{relay-report,txgen-summary}.json`.
| Measured, 3-node fast-time run (17:49 to 17:52 UTC) | Value |
|---|---|
| Sent to A / included / pending at the end / failures | 240 / 240 / 0 / 0 (0 nonce retries, 0 deferred, 0 throttled) |
| Included per second over the send span | 1.98 (target 2) |
| Inclusion latency p50 / p90 / p99 / max | 1,545 / 3,058 / 5,033 / 6,017 ms (mean 1,859) |
| Chain blocks in the window / executed transfers / skipped copies | 149 / 256 (240 transfers and 16 funding) / 0 |
| Included by miner B (one hop): blocks / with transactions / executed | 79 / 59 / 151 |
| Included by miner C (two hops): blocks / with transactions / executed | 70 / 42 / 105 |
| Pool depth, sampled every 5 s on A, B and C | equal on all three at 27 of 27 samples (0 to 6 pending), peak 6 |
| Sinks agree at the end | yes |
| First funding transfer, sent to A, included | 2.0 s after the send (block 37, mined by B or C) |
Reading. Every transaction given to A was mined by B or C within 6 s, two thirds of them within 3 s, with no skipped copy: the hold on block-added kept B's and C's parallel blocks from carrying the same transfer. The afternoon run on the live devnet, through one node with the cooldown, had p50 40.7 s and p90 110.8 s with 50-s quiet stretches; here the 1.5 s p50 is one fast-time block plus the relay and the executor's lag. The pool depth matching on all three nodes at every sample is the convergence. Not measured here: a transaction flood above the per-peer rate (the bucket is unit-tested only), a 13 peer in the fleet (the digest check and the version gate are the evidence), and the hold's 30-s expiry on a block that never reaches the chain (not seen in 149 chain blocks).
Commands: `IGNEUMD=vendor/igneum-node/target-txgossip/release/igneumd IGNEUM_MINER=vendor/igneum-node/target-txgossip/release/igneum-miner tools/lock/with-lock.sh run node tools/txgen/relay-net.mjs --rate 2 --duration 120 --wallets 16 --fund 2`; the Mac binaries from the fork worktree with `CARGO_TARGET_DIR=vendor/igneum-node/target-txgossip cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` under the build lock (an APFS clone of `target-036`, 2 min 15 s to clone, 5 min 06 s to build); the suites with `node tools/build-job.mjs run --target 1ccfe586 --node vendor/igneum-node-txgossip --targets linux --node-tests "igneum-exec kaspa-p2p-flows" --no-app`.
## 5 October 2026 (night), the SP1 CPU prover on PC 1 beside the miners, and the backend survey: no zkVM proves on AMD (amd-prove agent)
the project lead, 22:50 BST: "test proving on the amd card?" and "can we test proving on mac?". The analysis with the backend table and the tier consequences: `docs/analysis/amd-proving.md`. The survey (SP1 v6.8.1 and `dev` 318dd530 of 28 Sep 2026, RISC Zero, Jolt, OpenVM, ICICLE, sppark; every claim cites a file or page there): on 5 October 2026 no zkVM proves on an AMD GPU; Apple silicon has RISC Zero's shipped Metal prover and ICICLE's Metal backend; SP1, the prover here, is CPU-only off NVIDIA.
Machine: PC 1 (machine ae432dc7), Windows 11, WSL2 Ubuntu 24.04 as root, 16 cores and 46,994 MB visible to the VM, the Igneum Miner app 0.3.9 mining on the RTX 5090 and the RX 9070 XT throughout (the 5090 at 89% mean utilisation, 59 to 70% minimum, from a 1-s `nvidia-smi` sampler under the run: the job never touched a card). Signed `run` job `cpu-prove-pc1-small2` (`tools/amd-prove/pc1-cpu-prove.ps1`), 20:49:00Z to 20:59:49Z: the hosted package `igneum-prove-wsl2-pv1b.zip` (sha256 df50dee5...) built WITHOUT the `cuda` feature (6 s warm; the first job `cpu-prove-pc1-small` built it cold in 126 s), `--mode id` the pinned pair (shard `0x2b1a81cb...`, aggregator `0x474678f3...`, pinned 2026-10-05T16:20:38Z), `SP1_PROVER=cpu`, `--mode shard --shard 0` under `/usr/bin/time -v`. Log: `node tools/jobs.mjs cpu-prove-pc1-small2 --all`.
| Fixture | SP1 cycles | Setup s | Core prove s (bytes, verify s) | Compressed prove s (bytes, verify s) | Wall s | Peak RSS | CPU |
|---|---|---|---|---|---|---|---|
| block-56-transfers-3shards shard 0 (200 pgas, one transfer) | 315,479 | 22.75 (client 19.46, shard keys 1.85, aggregator keys 1.44) | 82.5 (7,310,257, 0.210) VERIFIED | 199.2 (1,272,897, 0.035) VERIFIED | 312.1 | 29,503,652 kB (29.5 GB) | 978% (9.8 of 16 cores), user 2,516 s, system 537 s, load max 11.3 |
| block-78-increment (2 transactions, 1 executed 1 skipped) | 631,127 | 21.75 | 87.0 (7,317,857, 0.209) VERIFIED | 202.3 (1,272,897, 0.034) VERIFIED | 322.3 | 30,517,916 kB (30.5 GB) | 979%, user 2,616 s, system 541 s, load max 13.1 |
| block-338-shard1 (one shard at `S_p`, 60.8 M cycles) | | not run: the PC 1 scheduler kept the machine for the Counter ASIC 2.0 gates (21:05Z). Approximate extrapolation: about 29 SP1 shards of 2^21 cycles at about 80 s each, 40 min of core proof, then hours of recursion; floor from the 5090's ratios (6x core, 4x compressed, block-78 to `S_p`): 9 min core, 13 min compressed | | | | | |
For comparison (this log): the Apple M5 Max CPU on 4 October, loaded, block-56 shard 0: core 83.1 s, compressed 272.3 s; on 3 October the v0 guest on block-78: core 22.0 s, compressed 55.7 s. The RTX 5090: block-78 core 1.4 s, compressed 2.7 s (4 October, mining paused); a full shard at `S_p` compressed 10.9 s alone and 33.0 s beside the miner; an empty live shard 7.0 to 7.7 s beside the miner (5 October). No fresh Mac run tonight: the measure lock was held from 20:31Z (a read-width `packbench`, three builds, a 1,500-s proving-v1 network under `run`) and did not free inside the 10-minute window set for it.
Reading, and the consequences (CLAUDE.md, every number). Doubling the cycles added 4.5 s to the core proof and 3.1 s to the compressed proof: about 280 s of a CPU proof is fixed cost in the compressed-proof recursion, so no shard size brings a CPU proof under the launch deadline (20 to 60 s behind the tip) or near the 10-s assignment window; it fits only the v1 unproven deadline (600 s), which pays a CPU prover only when no card has proven the shard in 10 minutes. The 29.5 to 30.5 GB peak RSS means the CPU prover needs 32 GB free: a 64 GB Windows PC (WSL2 takes half the host's RAM by default), a 32 GB Linux machine, a 64 GB Mac; a 16 GB machine cannot run it at all. Per tier: an AMD-only home miner (8, 12 or 16 GB, Windows or Linux) mines and does not prove, and loses the 20% proving-pool share; Apple silicon the same (the M5 Max mines at 26.7 MH/s, this log, 4 October); a mixed rig proves on its NVIDIA cards and the rig installer's `prover_decision` already skips every non-NVIDIA card (`packaging/linux/bin/igneum-rig-lib.sh`, branch `rig-install`), now a stated requirement; the app's `provedefault.rs` already keeps proving off on Apple silicon and off without an NVIDIA card. Decision asked of nobody: no CPU tier (the analysis, section 4a); the public line for the site, litepaper and Proving tile is in section 4c ("Proving needs an NVIDIA card with 16 GB or more today ... AMD and Apple cards mine. A prover for them lands when a zkVM ships one"). The first job proved nothing because an apostrophe inside a single-quoted awk program ended the quote and bash refused the loop while the job reported exit 0; the class fix is `tools/amd-prove/check-job-bash.sh` (`bash -n` on the embedded bash body before publishing) and the same `bash -n` inside the job before the run, both shown to refuse the bad body and pass the fixed one.
## Counter ASIC 2.0, the numbers
5 October 2026 (night). The chip-resistance layers measured on the three cards we own (Apple M5 Max, RTX 5090 on PC 1 and PC 2, RX 9070 XT on PC 1's eGPU), the decisions taken under the project lead's delegated rules for the devnet, and the chip model before and after. Every number is from an entry above or from the plan documents named; approximate is marked. Levels: `docs/plans/counter-asic-2-public.md`.
**Program class v3 (the devnet, activation by height switch `program_class_v3_activation_daa`)** = class v2's 128 x 4-byte loads, the era draw of the table layout and the working-set windows (layers 4 and 8), the cache growth rule (layer 6, option C: the cache doubles when the dataset doubles), the mixer at x8 (M16's multiplier), reserve family R1 (integer matrix, switched off) and the epoch length as a signalled reserve parameter (layer 9, 3,600 DAA s until a 90% signal). Not adopted on the measurements: wider reads (layer 1), the per-load width mix (layer 2), the per-warp write scratch (layer 3), the hot table (layer 5).
| Card | v2 MH/s | v3 MH/s, six eras (spread) | Bytes per hash | Latency-bound share | Daily 1 GiB build, v2 / v3 |
|---|---|---|---|---|---|
| Apple M5 Max, Metal | 27.68 | 27.85 to 27.98 (0.5%) | 512 | 1.06 | 21 / 21 ms |
| RTX 5090, CUDA | 137.2 | 135.90 to 137.70 (1.3%) | 512 | 1.01 | 25 / 23 ms |
| RX 9070 XT, OpenCL | 18.09 | 18.59 to 19.18 (3.1%) | 512 | 0.95 | 74 / 75 ms |
CPU verifier, one M5 Max core at load average 5.5 (the fixed crate, ca2-mixer 1ab8b21): v2 0.61 ms per warp, v3 (x8) 2.08 ms (3.4x), worst cold 2.15; the 10 ms gate holds 4.8x (4.6x on the worst cold unit). Bit-exact: every v3 pack's fingerprint equal on Metal, Apple OpenCL, CUDA and AMD OpenCL.
| Layer | Measured | Decision | The number |
|---|---|---|---|
| 1 wider reads | w16 139.8 / 17.90 / 28.26 MH/s (5090 / 9070 XT / M5 Max) against v2 136.1 / 18.15 / 27.74; w64 71.9 on the 5090 (share 0.58, 37% of its stream) | out: keep 4 B | the 9070 XT does 2.4 G dependent reads/s at every width; wider reads make the 5090 bandwidth-bound |
| 2 width mix per load | spread over six programs 18.8 / 7.4 / 11.3% and 22.3 / 5.5 / 8.1% | out | the 5% rule |
| 3 write scratch | GPU cost 12 to 48% at 32 and 128 KB per warp; the on-die-cache chip 2.4x at every share | out (the construct is sound; its tests stay) | the verifier resets the scratch per unit, so a chip keeps it in 80 to 320 B per lane |
| 4 and 8 era layout and windows | six-era spread 1.3 / 3.2 / 0.8% | in | under the 5% rule; the SRAM mirror a chip needs is the whole dataset every hour |
| 5 hot table | added form g 0.87 / 0.85 / 0.84 (5090), 0.84 / 0.81 / 0.80 (9070 XT) at 32 / 64 / 96 MiB | out (a 3.0 option) | no card keeps 32 MiB resident while the dataset streams; the replaced form helps the chip |
| 6 cache schedule | the 256 MiB mirror is 128 mm^2 and $46 at N5 by shipped cache-die density, approximate | option C, in | the cache's job is to stay above GPU L2 (96 MB on the 5090, 128 MB on GB202) |
| 7 integer matrix | dp4a 1.17x a step on the 5090, 1.06x on the 9070 XT, 1.6x emulated on Apple; mm8 native on all three as a tile | reserved R1, off | unlock at era 4 or 90% signal |
| M16 mixer | x4: verifier 1.24 ms, chip 1.84x with the allowance; x8: 2.08 ms, 0.92x; the daily build unmoved on every card | x8 in | the only lever that moves the named chip |
| 9 epoch length | compile-ahead 0.5 s (M5 Max, race off), 1.0 s (5090), 38 s with the race; FPGA compiles 42 to 160 min (PRflow, FPT 2019) | reserved, 600 s to 2 h by signal | at 600 s a per-program bitstream mines 0% of each epoch |
**The chip model, before and after** (`docs/analysis/chip-model-v3.md`, `docs/analysis/sram-mirror.md`): the strongest chip we can name holds the whole 256 MiB cache on-die (about 128 mm^2 and $46 of silicon at N5, approximate) and computes dataset items on the fly at 50 T integer op/s. Against the RTX 5090's measured 136.1 MH/s: class v2 333 MH/s, 2.4x; class v3 41.7 MH/s, 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon. The claim is "under 2x"; the margin is thin on the allowance (3.3x reads 1.0x) and 9% on the budget. Next levers, named: the mixer at x16 (the verifier at about 4 ms per warp; a 2019-class core unmeasured), a hot table small enough to stay resident beside the streaming dataset.
**The user tiers.** AMD RDNA 4 sits at about a seventh of a 5090 on this hash (its dependent-read rate: 2.4 G against 17.5 G per second), 2.2x worse per pound at list prices and 4.9x worse per watt (approximate); the card's memory system, not a tuning gap. The integrated tier on the CUDA and OpenCL one-click workers mines v3 with a restart per epoch until per-day dataset reuse lands (0.3.12). Card lifetime under the step schedule: a 4 GB card to year 4, 8 GB to year 12, 12 GB to year 28 with the cache freed after the daily build.
**The bounty.** A bounty for any chip design beating a GPU by more than 2x on the published model, with a leaderboard by card model, follows the external review (spec O-1.17, January 2027); it is named publicly only once escrowed (`docs/plans/funding.md`, rule 3), which it is not yet.
## 5 October 2026 (evening), proving v1: segment records, the chain rule, the unproven rule; what was measured tonight (proving engineer)
Branches `proving-v1` (main repository, worktree `igneum-wt-proving-v1`; fork `vendor/igneum-node-pv1` from a24ab01a). Rules: spec 7.8; plan `docs/plans/proving-v1.md`. Every row names its command. The live devnet was in a degraded state the whole evening: from 18:35Z the RTX 5090 workers on PC 1 and PC 2 exited at start on a pack seed mismatch (`the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT`, restart 60+ on PC 2 by 19:05Z, another agent's branch `pack-loop`), the Mac app node was down from 17:45Z, so PC 2 mined 3.4 MH/s from its iGPU and PC 2's prover was the only prover; the coordinator held every PC 2 measurement at 19:00Z until the fleet mines again.
### Step 1, the prover default and its cost
| What | Measured |
|---|---|
| The default rule (`app/igneum-app/src/provedefault.rs`) | `cargo test --release -p igneum-app provedefault` on this Mac (the app crate, build lock, 19:05Z): 5 passed (a 5090 with WSL2 on Windows is on; Windows without WSL2 off with the Set up hint; Linux needs no WSL2 and the 12 GB gate holds, a 10 GB 3080 and a 16 GB AMD card stay off; Apple silicon off; the biggest qualifying card is named) |
| Mining alone against mining with the prover, first try (PC 2 job `prover-cost-pc2-pv1`, `tools/proving-v1/pc2-prover-cost.ps1`, published 18:40:48Z, ran 18:41:13Z) | VOID: the job waited 20 min for the 5090 worker to hash and it never did (the pack fault above); "mining alone" was 0 MH/s |
| Mining alone against mining with the prover, the re-run after the coordinator's go (job `prover-cost-pc2-pv1b`, ran 19:20:36Z to 19:51:17Z; the 5090 worker restored at 19:16Z and hashing throughout; prover OFF by `POST /api/prove {"on":false}` 19:40:39Z, back ON 19:46:09Z, left on). The job's own `/api/state` samples stayed empty on PC 2 (`Invoke-RestMethod` returns an object PowerShell 5.1 cannot walk, `cards=0`, the fix is for the next run), so the hash rate is read from the miner's own STATUS lines (`miner-nvidia-1ccfe586-1` uploads, `now=... MH/s wall`, one every 30 s, the intake table `miner_logs`) | prover OFF, 19:41:09 to 19:46:09Z: n 10, mean 124.72 MH/s, p50 124.81, min 124.10, max 125.38. Prover ON, 19:46:39 to 19:51:13Z: n 9, mean 119.74, p50 118.87, min 118.08, max 123.42. The 15 min before the job with the prover on (19:25 to 19:40Z): n 30, mean 119.88, p50 118.79. So the prover costs the 5090 5.0 MH/s, 4.0% of its hash rate, while it proves the devnet's empty shards one after another (1.4 a minute here: the node's paidShards 559 -> 566 over the 5-min phase). A full shard at `S_p` keeps the card busier (the 4 October run proved one in 10.9 s); the cost at that load is the chain job's row |
| GPU memory during proving, first try (phase B of the first job: the prover on for 5 min, 298 one-second `nvidia-smi --query-gpu=memory.used` samples, the 5090 worker dead so the card held nothing else) | memory.used min 1,654 MiB, max 13,816 MiB, utilisation mean 2.7%, power max 190.6 W: the prover alone on empty shards |
| GPU memory with the miner AND the prover on the card (the re-run's phase B, 298 one-second samples, 19:46 to 19:51Z) | memory.used min 3,396 MiB (the miner's dataset and program resident), max 15,590 MiB, utilisation mean 92.9%, power max 328.6 W. So the prover's own peak is about 12.2 GB on an empty shard (15,590 minus the miner's 3,396), and the two together need 15.6 GB: a 16 GB card (5080, 9070 XT class, if it had a CUDA path) sits 0.4 GB under tonight's peak with no room for a full shard, a 24 GB 4090 has 8.4 GB of headroom, a 12 GB card cannot mine and prove at once on this build. The full-shard peak is the chain job's row |
| Shards per minute with the mining worker dead | the node's `paidShards` 510 -> 518 over the 5-min phase: 1.6 shards a minute from one 5090 through the app's loop (export, cut, prove, sign, submit) |
| Host RAM (Windows `Win32_OperatingSystem` and the `vmmem` working set, sampled every 15 s) | host used 25,550 MB of 63,132 MB at the end; the WSL2 VM's working set 7,915 MB (2,334 MB used of 30,914 MB inside the distribution) |
| The SP1 GPU server's compiled targets (`cuobjdump --list-elf /root/.sp1/bin/sp1-gpu-server` inside PC 2's Ubuntu-24.04, CUDA 12.8, driver 610.47) | `sp1-gpu-server` 6.8.1 (251,306,680 bytes, sha256 c2642ad1c42e85d8525159cf0c7cd5200d8766c9be1283f452a1f9bf9fea725c, the asset `sp1_gpu_server_v6.8.1_x86_64.tar.gz` the SDK downloads, `sp1-cuda-6.8.1/src/server.rs`): one ELF each for sm_80, sm_86, sm_89, sm_90, sm_100 and sm_120; `strings` finds compute_120 PTX as well. So sm_89 (Ada: RTX 4090, 4080) is compiled in natively, no JIT; so are Ampere (3090, 3060), Hopper, Blackwell datacentre (sm_100) and consumer (sm_120, the 5090). Nothing for AMD (no HIP path in SP1) |
### Step 2, aggregated chains
| What | Measured |
|---|---|
| The new host (`--mode chain`, `aggregate`, `verify-segment`) against every fixture natively | `igneum-prove-host <f> --mode native` on the Mac for the 12 fixtures of `proving/fixtures/` (9 block, 3 fee-switch), host built from this branch 19:06Z: every one MATCHES (the package gate's native half); `--mode id`: shard `0x2b1a81cb...`, aggregator `0x474678f3...`, the 0.3.9 pin, unchanged |
| Eight consecutive live fixtures | `igneum_exportSegments 0x0..0x13cb4` on node 1's exec RPC (127.0.0.1:26790, read-only, 20:06 BST, tip 81,076): 71,042,616 bytes, 81,077 segments, 28 accounts, 0.5 s; `igneum-prove-export export.json <n> block-<n>.json` for 81046..81053: replayed 81,077 segments from genesis in 1.8 s each, every state root equal to the node's; one shard a block, 0 pgas (no transactions on the devnet tonight), `proving/fixtures/chain/` |
| Chain of 2 on the Mac CPU (the known-finished case of `--mode chain` before the GPU; M5 Max under the live nodes, the harness and two builds) | `SP1_PROVER=cpu igneum-prove-host --mode chain --chain block-81046.json,block-81047.json --out results.json` under the run lock, 19:07:48Z to 19:11:28Z: setup 12.2 s; block 81046: shard 0 compressed 55.4 s (1,272,897 bytes, verify 0.036 s), aggregate 52.0 s (1,272,909 bytes, verify 0.031 s), chain_len 1, agg_vk zero; block 81047: shard 41.3 s, aggregate WITH the previous block proof 59.1 s, chain_len 2, agg_vk = the pinned aggregator id; end to end 207.9 s; final proof 1,272,909 bytes, statement 0x232276f4... The recursion over the previous proof cost 7 s more than the first aggregation on this CPU |
| `--mode verify-segment` on that proof (the node's path: SP1 light verifier, pinned aggregator key) | VERIFIED in 0.032 s (0.27 s wall, three runs: 0.033, 0.032, 0.032); known-failed: a wrong statement NOT VERIFIED (0.032 s); the shard verifier (`--mode verify`) on the segment proof NOT VERIFIED, "program id 0x474678f3... IS NOT OURS 0x2b1a81cb..." |
| Chain of 8 on the RTX 5090 (N = 2, 4, 8), job `chain-pc2-pv1b` (`tools/proving-v1/pc2-chain.ps1`; the package `igneum-prove-wsl2-pv1.zip` eb6dccf8..., 1.5 MB, fetched by `fetch-prove-pv1` 19:51Z; the first try `chain-pc2-pv1` died in its own export step, fixed) | Ran 19:58:37Z: the export from PC 2's node (72,901,414 bytes, 1.4 s), the host built in WSL2 against the live build's warm target dir in 6 s and installed to `/opt/igneum-pv1` (the live `/opt/igneum` host untouched, sha 29cc4768...), `--mode id` the pinned pair; eight consecutive fixtures 83346..83353 cut, every one MATCHES natively. The chain on the GPU (SP1_PROVER=cuda, the miner mining on the same card at 119 MH/s): setup 12.7 s; block 83346: shard 7.4 s, aggregate 7.6 s (chain_len 1), 15.1 s; block 83347: shard 7.2 s, aggregate WITH the previous proof 9.5 s (chain_len 2, agg_vk the pinned aggregator id), 16.8 s, cumulative 31.8 s over 2 blocks; block 83348: shard 7.0 s, then at 20:01:09Z the app quit and aborted the job ("quit: stopping the miners, then the node", then "job chain-pc2-pv1b: aborted (the app is quitting)"; NOT an update: nothing of 0.3.10 was published; the log gives the quit no source; 20 s earlier the efficiency sweep's administrator prompt had been cancelled at the keyboard, and 13 s earlier the live prover had failed with "CudaClientError: Connect(PermissionDenied)", the root-owned socket my job had left, below). So N = 2 measured: 31.8 s of GPU time for two empty blocks, the chained aggregation 1.9 s dearer than the first; N = 4 and 8 are the re-run `chain-pc2-pv1c` after the restart. An empty shard's compressed proof on the 5090 is 7.0 to 7.4 s (the 200-pgas shard of 4 October took 2.7 s with the card to itself; tonight the miner held it at 92% utilisation) |
| The chain of 8, the third run `chain-pc2-pv1c` (20:05:21Z to 20:08:33Z, after the app restart; blocks 83616..83623 from PC 2's node at tip 83646, the same script; results `tools/proving-v1/chain-pc2-2026-10-05.json`) | Build 5 s (warm), eight fixtures cut and MATCHING natively, setup 15.7 s, then on the GPU with the miner mining on the same card: shard proofs 7.3 to 7.7 s each (8 x, 59.5 s), aggregations 7.9 s for the first block and 9.6 to 9.7 s for every chained one (75.5 s), every proof VERIFIED, end to end 135.6 s for 8 blocks (17.0 s a block from the second on). Cumulative: N = 2 at 32.6 s, N = 4 at 66.8 s, N = 8 at 135.6 s. The final proof is 1,272,909 bytes whatever N (chain_len 8, agg_vk the pinned aggregator id), the record 586 bytes; `--mode verify-segment` on it: VERIFIED in 0.039, 0.037, 0.040 s after a 0.26-s light-verifier setup, the same three runs each time. GPU memory over the chain (152 one-second samples): max 16,751 MiB with the miner's 3.4 GB resident, so the chained aggregation holds about 13.4 GB, 1.2 GB over the shard-only peak; WSL used 2,456 MB |
Reading the chain numbers. Aggregation is a fixed cost per block (9.7 s here), not per segment: the recursion verifies one more proof whatever `chain_len`, so the record for N blocks costs N aggregations and the verifier one. Against 4 October with the miner stopped (aggregate 2.2 to 2.5 s, a 200-pgas shard 2.7 s), tonight's 9.7 s and 7.3 s say the miner's 92% utilisation slows the prover about 3 to 4x while the prover slows the miner 4%: the card is shared, and the lottery wins the arbitration. A machine that mines and proves at once delivers one empty block's proof and aggregation in 17 s; one that only proves, about 5 s (approximate, from the 4 October stages).
### Step 3, coverage
| What | Measured |
|---|---|
| A 3-minute window at 18:57Z on node 1 (`node tools/proving-v1/coverage.mjs --minutes 3`, chain blocks 80754..80839, 86 blocks) | 4 blocks with a paid shard (4.7%), 4 fully proven, 4 of 86 shards; on-chain latency (carrier timestamp minus block timestamp) n 4: min 36 s, p50 39 s, max 44 s; 0 content blocks. One prover (PC 2), the Mac verifier node down, PC 2 producing few blocks (3.4 MH/s): the degraded state above, not the fleet's number |
| A 30-minute window, 19:13 to 19:43Z, the degraded fleet (PC 2 the only prover, its 5090 worker restored at 19:16Z, the Mac app node down by decision: the Mac app is attached to node 1) | `node tools/proving-v1/coverage.mjs --minutes 30 --watch` on node 1: chain blocks 81236..82668, 1,433 blocks; 38 with a paid shard (2.7%), all 38 fully proven (one shard a block, 0 content blocks); on-chain latency n 38: min 36, p50 44, p90 52, p99 62, max 65 s. The live page's 10-minute proving object read 0 shards and 0 provers at 19:42Z (it counts what its own node verified; that node is the Mac app node, down), so the chain's own count is the number |
| A 30-minute window with the fleet mining (PC 2 at 119 MH/s from 19:16Z, PC 1 at 128.8 from 19:18Z; PC 2 still the only prover, its prover OFF for the 5 min of the cost job's phase A inside this window; the Mac app node down by decision) | `coverage.mjs --minutes 30 --watch`, 19:21 to 19:51Z on node 1: chain blocks 81644..83069, 1,426 blocks; 34 with a paid shard (2.4%), all fully proven (one shard a block, no content); on-chain latency n 34: min 38, p50 44, p90 51, p99 52, max 53 s. One 5090 through the app's loop as it is covers 2.4 to 2.7% of the blocks; the latency from block to carried record is 44 s at the median, under the litepaper's minute, and would be the same for every block if the fleet were 40 cards (the table below) |
### Step 3, the fleet size (arithmetic from measured inputs; every input names its entry)
Inputs, all RTX 5090 (PC 2), SP1 6.8.1 cuda: a full shard at the provisional `S_p` (6.75 M pgas) compressed in 10.9 s and the four shards of a near-`B_p` block in 10.2 to 10.7 s each (bench-log 4 October 2026, "shard proving on the RTX 5090", runs run-20261004-173115 and run-20261004-r3-shards); one aggregation 2.2 s (two shards) to 2.5 s (four shards), the same entry; tonight's chain of 2 on the Mac CPU shows the recursion over the previous block proof costs the same order as a first aggregation (52.0 s against 59.1 s), so the GPU figure for a chained aggregation is taken as 2.5 s, approximate, until the held PC 2 chain job measures it; the app's live loop tonight: 1.6 shards a minute per card on empty shards (export, cut, key setup, prove, sign, submit: about 37 s a shard, of which the proof is a few seconds), bench-log step 1 above. A 5090 proves one thing at a time.
| Block content at 1 block/s | Shard proofs a second (fleet) | Card-seconds a second for shards | Aggregations a second | Card-seconds a second for aggregation | 5090-class cards for 100% | Rule |
|---|---|---|---|---|---|---|
| empty blocks (tonight's devnet), the app's loop as it is, the card also mining | 1 | 37 | 1 | 9.7 (measured, `chain-pc2-pv1c`) | 47 | one shard per block, the loop's 37 s each plus a chained aggregation |
| empty blocks, the chain mode's shape (one key setup per process, proofs back to back), the card also mining | 1 | 7.4 (measured) | 1 | 9.7 (measured) | 18 | 17.1 card-seconds a block, `chain-pc2-pv1c` |
| empty blocks, cards that only prove | 1 | 2.7 (4 October, a 200-pgas shard) | 1 | 2.5 (4 October) | 6 (approximate) | the miner's 92% utilisation costs the prover 3 to 4x |
| one full shard a block (`S_p`, 6.75 M pgas), cards that only prove | 1 | 10.9 | 1 | 2.5 | 14 | 4 October's stages |
| one full shard a block, the card also mining | 1 | about 35 (approximate: 10.9 x 3.2, tonight's ratio) | 1 | 9.7 | about 45 (approximate) | the full-shard proof with the miner on the card is not measured |
| blocks at `B_p` (four full shards), cards that only prove | 4 | 42.5 | 1 | 2.5 | 45 | 4 x 10.6 + 2.5 |
| at the adopted v1 budgets (`B_p` 120,000 pgas, `S_p` 30,000, from DAA 210,000 on the devnet): a v1 shard of transfers ran at 213 to 236 cycles per pgas (bench-log 5 October, "the prover carries both fee tables"), 7 M cycles a shard against 60 M for the prototype shard | 4 | under 42.5 (the 5090 time for a 7 M-cycle shard is not measured; scaling 10.9 s by cycles gives about 1.3 s, approximate) | 1 | 2.5 to 9.7 | 8 to 15 (approximate) | measure before the switch lands |
Reading. The card count is the sum of card-seconds of work per block-second, rounded up, with no slack for the exclusive window, the relay or a card's idle gaps; the devnet's own numbers tonight (one card, 1.4 to 1.6 shards a minute, 2.4 to 4.7% of blocks) are the first row. Two levers, both measured tonight: the loop (a shard's carriage through export, cut and a 12-s key setup is 25 s on top of a 7-s proof; the host's `--mode aggregate` and `--mode chain` hold one key setup per process and the prover loop should do the same, the 0.3.11 item in the plan) and the card's other job (a mining card proves 3 to 4x slower than an idle one, `chain-pc2-pv1c` against 4 October; the prover's cost to mining is 4%). A fleet of 18 mining 5090s, or 6 proving-only ones, covers an empty-block chain at 1 block/s through the chain mode; the mandatory rule waits for the measured share to reach one, not for these rows.
### The 12 GB requirement (the project lead, 20:1xZ: "make sure we can prove on 12gb cards"): the GPU memory peak against SP1's knobs
Job `memsweep-pc2-pv1` (`tools/proving-v1/pc2-memory-sweep.ps1`), PC 2's RTX 5090 (32,607 MiB), the miners STOPPED by the job and the live prover switched off (its `sp1-gpu-server` would otherwise be the one the client connects to), every row: the server killed first, a 1-s `nvidia-smi memory.used` sampler, one `--mode compressed --shard 0` run of the pv1 host (`/opt/igneum-pv1`, SP1 6.8.1 cuda, `sp1-gpu-server` 6.8.1), 20:19 to 20:25Z. The knobs are the environment the GPU server inherits from the host process (`sp1-core-executor-6.8.1/src/opts.rs`: `SHARD_SIZE`, `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `MINIMAL_TRACE_CHUNK_THRESHOLD`, `TRACE_CHUNK_SLOTS`; `sp1-prover-6.8.1/src/worker/config.rs`: the `SP1_WORKER_NUM_*` and `*_BUFFER_SIZE` counts, defaults 4 core workers, 8 recursion prover workers). Idle card before the sweep: 1,732 MiB.
| Config (environment) | Fixture | Cycles | Peak MiB | Compressed prove s | Verified |
|---|---|---|---|---|---|
| baseline (no knob) | block-338-shard1, a full shard at `S_p` (6.75 M pgas) | 60,415,376 | **28,295** | 11.4 | yes |
| baseline | block-83616, an empty live shard | 280,706 | **13,863** | 2.3 | yes |
| ELEMENT_THRESHOLD 2^27 | full shard | 60.4 M | 28,326 | 10.9 | yes |
| ELEMENT_THRESHOLD 2^26, HEIGHT_THRESHOLD 2^21 | full shard | 60.4 M | 28,326 | 10.7 | yes |
| every worker count and buffer 1 | full shard | 60.4 M | 28,326 | 20.8 | yes |
| every worker count and buffer 2 | full shard | 60.4 M | 28,327 | 12.9 | yes |
| workers 1 + ELEMENT 2^27 | full shard | 60.4 M | 28,263 | 20.3 | yes |
| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 | full shard | 60.4 M | 28,326 | 20.6 | yes |
| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 + trace chunks 4 M x 2 slots | full shard | 60.4 M | 28,358 | 22.6 | yes |
| workers 1 + ELEMENT 2^25 + HEIGHT 2^20 | full shard | 60.4 M | 22,919 | 22.2 | yes |
| workers 1 + ELEMENT 2^26 + HEIGHT 2^21 | empty shard | 280,706 | 13,861 | 2.6 | yes |
Reading. The GPU memory of a compressed shard proof is **13.9 GB for a shard of 280,000 cycles and 28.3 GB for one of 60 M cycles**, and no knob the environment carries moves the floor: the worker counts only slow the proof (11.4 s to 20.8 s), the trace thresholds at 2^26 and 2^27 change nothing, and the smallest trace threshold tried (2^25 elements, 2^20 rows) takes 5.4 GB off the full shard (22.9 GB) at twice the time. The floor sits in the GPU server's own allocation, not in the shard: an empty shard with every knob at its minimum still takes 13.9 GB. So on SP1 6.8.1's `sp1-gpu-server` as shipped, **a 12 GB card cannot prove even an empty shard** (13.9 GB), and the 11.0 GB target of tonight's requirement is out of reach from the environment. The S_p/2 and S_p/4 cuts of block 344 did not run: the package carries no `tools/prove-fixtures/seq.json` (the cut rows need the export; they would sit between the two measured points, and the floor is the binding number anyway). What is left to try, in order: the server's own options (its `--help` and the option names in its strings: the miner-on job prints them), SP1's core-only proof (the node needs the compressed proof, so this changes the protocol), and an SP1 release built for smaller cards (the 6.8.1 release notes are not read here; approximate: the project's documentation names 24 GB as the GPU requirement, `proving/windows-wsl2/setup-wsl.sh` quotes it).
### The same shard with the miner running (the 16 GB requirement), and the GPU server's own options
Job `memminer-pc2-pv1` (`tools/proving-v1/pc2-memory-miner-on.ps1`), 20:28 to 20:30Z, the miner at full rate on the card, the live prover off for the run, the same 1-s sampler: the full shard at `S_p` (60.4 M cycles) peaked at **30,039 MiB** and took 33.0 s (28,295 MiB and 11.4 s with the card to itself: the miner costs the prover 2.9x in time and 1.7 GB of memory); the empty shard **15,670 MiB** and 7.7 s (13,863 and 2.3 s alone). So a 32 GB card mines and proves the prototype shard with 2.5 GB to spare; a 24 GB card cannot prove it even alone (28.3 GB); a 16 GB card cannot hold even the empty shard beside the miner (15.7 GB, the display and driver on top). `sp1-gpu-server --help` prints only `--version`: it has no options of its own, and its strings carry no memory setting (`CUDA_OUT_OF_MEMORY` is an error name). The shard SIZE is therefore the only lever left on this build, measured next as the S_p curve.
The root-socket fault (the class, fixed the same evening). The chain and memory jobs ran the host as root inside WSL2; the first `sp1-gpu-server` they started left `/tmp/sp1-cuda-0.sock` owned by root, and the live prover (the app's own WSL user) then failed every shard with `CudaClientError: Connect(Os { code: 13, kind: PermissionDenied })` (PC 2 app log 1791230456, 20:00:56Z) until the socket was gone. Every pv1 playbook now kills the server and unlinks `/tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` fails CI on any playbook that runs a prove mode as root without both lines (shown failing on `pc2-prover-cost.ps1` before its `--mode id`-only exemption, passing after), and the plan carries the rule: a prover job on a shared card runs as the app's user or cleans its socket. It recurred at 21:25Z from another agent's job (agg-cost-pc2-1, the same root-run shape) and survived the 0.3.10 restart at 21:49:41Z; the fix job `socketfix-pc2-pv1` (`tools/proving-v1/pc2-socket-fix.ps1`, 22:01:14 to 22:02:12Z) found `/tmp/sp1-cuda-0.sock` owned by root, removed it, switched the prover off and on, and the app's next shard (block 89011 shard 0) was proven and submitted in 34 s and paid 0.93 IGN at 22:02:24Z. Playbooks that run the host: `tools/proving-v1/pc2-chain.ps1`, `pc2-memory-sweep.ps1`, `pc2-memory-miner-on.ps1`, `pc2-sp-curve.ps1` (all root, all with the cleanup now; the first two chain and sweep runs had none), `pc2-prover-cost.ps1` (`--mode id` only), `relay/playbooks/shard-test.ps1` and `proving/windows-wsl2/prove-shard.sh`, `prove-block.sh` (the app's user, not root), `tools/proving-v0/run.mjs` (the Mac, no server).
### The S_p curve: peak GPU memory against shard size against time, the card to itself (the first of the two curve jobs)
Job `spcurve-stopped-pc2-pv1` (`tools/proving-v1/pc2-sp-curve.ps1`, the miners stopped by the job, the live prover off, the server killed and its socket unlinked around every point, a 1-s `nvidia-smi` sampler), 20:33 to 20:37Z, PC 2's RTX 5090, the pv1 host (this run's `--budget` points were ignored by the pv1 host, so its block-344 rows are the fixture's own 6.75 M-pgas shard 0 twice; the pv1b host's re-plans at 2.25 M and 4.5 M pgas are the next job's rows). Idle card 1,743 MiB.
| Shard | pgas | Witness bytes | SP1 cycles | Peak MiB, card alone | Compressed prove s | Knob |
|---|---|---|---|---|---|---|
| block 83616, an empty live shard | 0 | 13,964 | 280,706 | **13,874** | 2.2 | none |
| block 56, one transfer | 600 | 4,902 | 556,369 | 13,907 | 3.2 | none |
| fees-v1-shards2 shard 0, a shard at the ADOPTED v1 budget (`S_p` 30,000; 4 transactions, 2 shards a block) | 22,172 | 18,390 | 4,717,439 | **20,434** | 4.3 | none |
| the same | 22,172 | 18,390 | 4.7 M | 20,435 | 3.7 | ELEMENT_THRESHOLD 2^25, HEIGHT 2^20 |
| block 338 shard 0, the full PROTOTYPE shard (`S_p` 7.5 M) | 6,751,568 | 21,611 | 60,415,376 | **28,307** | 10.8 | none |
| the same | 6.75 M | 21,611 | 60.4 M | 22,963 | 11.5 | ELEMENT_THRESHOLD 2^25, HEIGHT 2^20 |
| block 344 shard 0 (the fixture's own cut, 6.75 M pgas, modexp) | 6,748,392 | 18,535 | 59,678,420 | 28,275 and 28,307 | 11.5 and 11.0 | none |
The second job (`spcurve-stopped-pc2-pv1b`, the pv1b host whose `--budget` re-plans a fixture, 20:43 to 20:47Z, the same conditions) repeats the points (empty 13,875 MiB 2.1 s; one transfer 13,907 MiB 3.3 s; the v1 shard 20,435 MiB 4.2 s; the prototype shard 28,275 MiB 11.2 s) and adds the re-plans of block 344 (27 M pgas of modexp): at 2.25 M pgas (one transaction, 16 shards a block, 19,987,938 cycles) **28,371 MiB** and 6.6 s; at 4.5 M pgas (7 shards a block, 40,011,108 cycles) 28,307 MiB and 8.5 s; with the 2^25 trace threshold the 2.25 M shard 22,835 MiB and 6.2 s. So the peak is flat at 28.3 GB from 20 M cycles to 60 M (the server's buffers step up between 4.7 M and 20 M cycles and not after), and cutting the prototype shard smaller buys nothing until the v1 size.
The third job (`spcurve-miner-pc2-pv1`, the same points WITH THE MINER RUNNING on the card, 20:49Z on, the live prover off): empty shard 15,585 MiB and 7.5 s; one transfer 15,745 MiB and 12.7 s; **the v1 shard 22,210 MiB and 13.2 s** (20,435 and 4.2 s alone: the miner adds 1.8 GB and 3.1x); the 2.25 M shard 30,049 MiB and 17.9 s; the 4.5 M shard 29,954 MiB and 26.3 s; the prototype shard 30,083 MiB and 33.3 s. With the 2^25 trace threshold beside the miner: the 2.25 M shard 24,642 MiB and 21.5 s, the prototype shard 24,739 MiB and 38.8 s (24.7 GB: over a 24 GB card by the display's share, and 3.6x slower than the card alone). So beside the miner the adopted shard needs 22.2 GB: a 24 GB card (24,564 MiB) has 2.3 GB spare for it (the number for a 24 GB card is the 5090's allocation pattern on a 32 GB card, so approximate for the card itself), and the prototype shard needs 30.1 GB, the 32 GB card alone.
Reading, with the miner-on pairs above (empty shard 15,670 MiB, full prototype shard 30,039 MiB). The witness is never the binding term (4.9 to 21.6 KB a shard); the GPU server's working set is: a floor of 13.9 GB for any shard, 20.4 GB at 4.7 M cycles, 28.3 GB at 60 M cycles (23.0 GB with the smallest trace threshold, at the same time). By card: a **12 GB card proves nothing** on this build (the floor is 13.9 GB alone); a **16 GB card proves only empty and near-empty shards, alone** (13.9 GB; 15.7 GB beside the miner leaves nothing for the display); a **24 GB card proves the adopted v1 shard alone** (20.4 GB) and, at the miner's measured 1.7 GB extra, about 22.1 GB beside it (approximate: not measured on a 24 GB card), and never the prototype shard (28.3 GB); a **32 GB card proves the prototype shard beside the miner with 2.5 GB spare** (30.0 of 32.6 GB). The devnet is on the prototype table until H = 210,000 (6 October, about 19:50Z) and on the adopted v1 table (`S_p` 30,000 pgas) after it, so from H the 24 GB tier joins the provers and the shard that binds the memory is the 4.7 M-cycle one. Shards per block at each size: 1 at the prototype `S_p`, 4 at `B_p`; at the v1 budget 1 to 4 (one a block on tonight's chain, 2 to 3 on the txgen blocks).
### Step 4, the rule
| What | Measured |
|---|---|
| Unit tests | `cargo test --release -p kaspa-consensus-core -p igneum-exec --lib -- proving config::params::tests::override_params_carry_the_proving_v1 config::params::tests::consensus_digest` on this Mac (target `vendor/igneum-node/target-pv1`, 19:09Z): consensus core 13 passed (the segment record round trip, signature and the three nested sections; the credit split; the params switch and the digest that moves only once the switch is set), exec 8 passed (the segment grid and the split; the record checks: alignment, block, chain length, the veto naming the field, the deadline, the window; the chain rule both ways; the unproven restart; the shard side at 90%; the pool offering the segment section). The six full node suites go to PC 2 as a build job when the fleet is back |
| The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed. Run 3 on the FINAL fork tree (ece42979 on the 0.3.10 commit 21d4c73c, protocol 15, N = 8 both in the params default and `--segment 8`, the fast-time file's four fields), 20:52:41Z to 20:56:45Z: PASSED, 21 checks in 244.4 s (segments of 8: 119..126 paid in 1.0 s after submission, 127..134 refused fresh and paid continuing with chain_len 16, 135..142 left unproven and skipped, 143..150 restarted the chain) |
## 5 October 2026 (night), dp4a-class throughput on the M5 Max: the dot4 emulation against the ALU chain (Counter ASIC 2.0 layer 7)
Apple M5 Max, macOS 26, branch `ca2-analysis` (base `readwidth` 4badcee). The probes are standalone (no pack, no lottery kernel): `proto-metal/dot4-probe.swift` (built `swiftc -O -o dot4-probe dot4-probe.swift -framework Metal` under `with-lock.sh build`), `proto-opencl/dot4-probe.c` (built `cc -std=c99 -O2 -o dot4-probe-cl dot4-probe.c -framework OpenCL`), both run under `with-lock.sh measure` (exclusive; nothing else built or measured on the Mac during the runs). Shape: a dependent chain of one dot4 per step per lane, `acc = dot4(x, y, acc); x = x * 0x9E3779B1 + acc; y = rotl(y, 7) ^ (acc + s)`, 1,048,576 lanes x 4,096 steps, work-group 256, best of 3 with a fresh seed per repetition, device time (Metal: command buffer GPU start to end; OpenCL: event profiling). Beside it the ALU chain of the 9070 XT entry (`x = x * K + rotl(y, 7); y = (y ^ x) + s`, 5 ops per step counted). Every kernel is checked bit for bit against a CPU reference on lanes 0 and 1,048,575 in every repetition ("ok"). Design context: `docs/analysis/int8-matrix-family.md`.
| API, kernel | What one step is | best ms | G steps/s | ns per dependent step | ok |
|---|---|---|---|---|---|
| Metal, `probe_alu` | mul, add, rotate, xor, add | 4.882 | 879.8 (about 4.4 T int ops/s at 5 per step, approximate) | 1,192 | yes |
| Metal, `probe_dot4s` | signed dot4 emulated: `int4(as_type<char4>(a))` x same for b, 4 products summed into a wrapping int, plus the 3-op chain | 22.820 | 188.2 G dot4/s | 5,571 | yes |
| Metal, `probe_dot4u` | unsigned dot4 emulated: `uint4(as_type<uchar4>(a))`, same chain | 7.834 | 548.2 G dot4/s | 1,913 | yes |
| Apple OpenCL 1.2, `alu` | as Metal | 4.928 | 871.5 | 1,203 | yes |
| Apple OpenCL 1.2, `dot4e` | signed dot4 emulated with `convert_int4(as_char4(a))` | 22.797 | 188.4 G dot4/s | 5,566 | yes |
| Apple OpenCL 1.2, `dot4_khr` | `acc + dot(as_char4(x), as_char4(y))` under `#pragma OPENCL EXTENSION cl_khr_integer_dot_product : enable` | 5.076 | 846.2 | 1,239 | NO: mismatched the CPU reference on every lane checked in all 3 repetitions |
Reading: on this GPU a signed-byte dot4 costs 4.7 ALU-chain steps and an unsigned-byte one 1.6; Metal has no dp4a and no integer simdgroup matrix (MSL 4.1 sections 2.4 and 6.9), so these are the honest Apple costs of a per-lane dot4 family, and an unsigned definition is 3x cheaper for Apple at no cost to NVIDIA or AMD (both carry the unsigned form, PTX `dp4a.u32.u32`, AMD `v_dot4_u32_u8`). Apple's OpenCL does not list `cl_khr_integer_dot_product`; its `dot` on `char4` compiled anyway and returned something other than the integer dot (the mismatch), which is why a family's conformance vectors must gate every vendor path on the feature macro, not on "it compiled". Not run here: NVIDIA and AMD. The PC job is prepared and not published (coordinator's rule): `relay/playbooks/dot4-probe.ps1` with `dot4-probe-cl.exe` (proto-opencl/dot4-probe.c cross-compiled with mingw as `x86_64-w64-mingw32-gcc -std=c99 -O2 -static -DIGNEUM_CL_DYNAMIC -DCL_TARGET_OPENCL_VERSION=120 -I proto-cuda/nvrtc/redist/include`, sha256 `5adaeb1aceb03dc41135baabe0b53f1ed5fac891a5b3c3849645b03efe4416f4`, 161,863 bytes); it runs the scalar, KHR, AMD `__builtin_amdgcn_sudot4` and NVIDIA inline-PTX `dp4a` variants on every OpenCL GPU of the machine with the mining cards switched off through `/api/cards` and restored after. The CUDA form (`proto-cuda/dot4-probe.cu`, `__dp4a`) needs nvcc on the PC and is the cross-check.
**PC 1, 5 October 2026 20:29 UTC, the same probe on the RTX 5090 and the RX 9070 XT** (machine ae432dc7, Windows 11; fetch job `fetch-dot4-20261005` placed `dot4-probe-cl.exe` sha256 `5adaeb1a…6416f4`, run job `run-dot4-20261005` ran `relay/playbooks/dot4-probe.ps1`: the app's `nvidia:0` and `amd:1:gfx1201` cards switched off through `POST api/cards`, the probe run on every OpenCL device, the cards restored with their settings (identities 8 and 2, power cap 80% and none); `node tools/jobs.mjs run-dot4-20261005`; 101 s wall, every kernel under 10 ms; device event time, best of 3, same lanes and steps as the Mac rows):
| Device, platform | alu, G steps/s (ms) | dot4e signed emulation, G dot4/s (ms) | dot4 instruction, G dot4/s (ms) | `cl_khr_integer_dot_product` | ok |
|---|---|---|---|---|---|
| RTX 5090, NVIDIA OpenCL 3.0 CUDA, driver 617.14 | 8,753.5 (0.491) | 1,239.1 (3.466), 7.1x the ALU step | 7,453.6 (0.576) via inline PTX `dp4a.s32.s32`, 1.17x the ALU step | not listed; the `dot(char4,char4)` kernel does not build | yes |
| RX 9070 XT (gfx1201), AMD-APP 3683.0 (PAL,LC), OpenCL 2.0 | 701.4 (6.124) | 480.8 (8.932), 1.46x | 664.3 (6.465) via `__builtin_amdgcn_sudot4`, 1.06x | not listed; same | yes |
| RX 9070 XT, the older 3652.0 platform entry (dup) | 696.2 (6.169) | 501.7 (8.561) | 683.6 (6.283) | not listed | yes |
| gfx1036 (integrated RDNA 2, 2 CUs), 3683.0 | 40.6 (105.9) | 15.8 (272.3), 2.6x | `sudot4` does not build: "needs target feature dot8-insts" | not listed | alu and dot4e yes |
Reading: one `dp4a` on the 5090 costs about one ALU-chain step (7.45 T dot4/s, 0.85 of the chain's 8.75 T steps/s); one `v_dot4_i32_iu8` on the 9070 XT the same (0.66 T, 0.95 of its chain). Emulating the signed dot4 costs 6.0x the instruction on NVIDIA (the OpenCL compiler does not fold the four sign-extended products into `dp4a`) and 1.38x on AMD. Vendor ratios: the 5090 is 12.5x the 9070 XT on the ALU chain and 11.2x on hardware dot4; against the M5 Max's best (unsigned emulation, 0.55 T) it is 10x on the chain and 13.6x on dot4. The hash itself is bound by DRAM reads, so these per-op numbers bound a family's cost and are not hash rates (`docs/analysis/int8-matrix-family.md` section 4). Adrenalin's OpenCL C accepts the clang builtin and emits the instruction on RDNA 4 (the third-party RDNA 3 report of the same route is now confirmed on this card); no PC platform lists the Khronos integer-dot extension. The 5090 SM clock read 2,505 MHz before and after (nvidia-smi; 2,850 MHz while mining in the telemetry entry), so the card was idle for the probe.
## 5 October 2026, layer 3 scratch soundness (Counter ASIC 2.0 step 4; branch ca2-soundness on readwidth b970dda; cryptographer)
Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0, other agents' builds and the readwidth measurements running beside (the Mac measure lock was free during the GPU runs; nothing here is a hash-rate figure). Write-up `docs/analysis/scratch-soundness.md`; tests `igneum-pow/tests/scratch.rs`; harness `proto-metal/packbench` built from this branch (`--batch-base` added) into the session scratchpad with `swiftc -O -target arm64-apple-macos11 -framework Metal`.
CPU, `with-lock.sh build nice -n 19 ~/.cargo/bin/cargo test -j4 --test scratch -- --nocapture` (3.6 s): 7 of 7 pass. Stats, 6 classes x 3 seeds x 2^11 units, every read-modify-write traced (3.1 to 12.6 million per class): written-word bias within 6 sigma (worst 3.63); re-hit rate measured against the uniform birthday rate 12.58 vs 10.91 percent (scr2k32), 21.47 vs 20.83 (scr4k32), 36.99 vs 36.50 (scr8k32), 3.84 vs 2.88 (scr2k128), 6.25 vs 5.82 (scr4k128), 11.89 vs 11.37 (scr8k128); slot histogram non-uniform (hottest slot 1.39x to 5.10x the mean: the slot is a register's low bits); deepest chain 5 to 9. Edge: 7 hand-built programs x 2 geometries x 4 bases against an independent hand model, 56 of 56, and 56 of 56 mismatches with the hand model's rewrite words swapped. Static scratch check: 42 of 42 emitted kernels of the 7 scr packs (regenerated byte for byte from program.json first), 6 deliberate breaks caught. Fuzz: 200 generated scratch programs, contract and acceptance on every instruction, 800 units; `IGNEUM_SCRATCH_PACKS_OUT` wrote 214 packs (57 s, three memory-hard caches). The crate's other tests: 33 of 34 lib tests pass; `verify::tests::fold_and_wide_fetch` fails on the readwidth tip itself (`verify.rs:508`, `k as u32 * 0x9E37_79B1` overflows under the test profile; not touched here).
Metal, `with-lock.sh run <script>`, scripts `gpu-a.sh` and `gpu-b.sh` in the session scratchpad (one `packbench` call per line):
| Run | Command shape | Result |
|---|---|---|
| scr4k32 standard pack, timing | `packbench --pack proto-cuda/packs-readwidth/scr4k32 --batches 1 --batch-log2 24 --warps 2048` | 3/3 standalone, 3/3 in batch, fingerprint `3d1af881bd978fb9`, 1.8 s wall for the run |
| warp-count independence | same pack, `--batch-log2 12 --warps 1`, `2`, `128` | fingerprint `8c07620f4d9adefd` at all three |
| wrap inside the launch | `--batch-log2 9 --batch-base 4294967040 --warps 1`, `4` | fingerprint `8e9e233234d3a297` at both, the base-0 vector inside the window after the wrap 1/1 |
| broken tag (`tag = salt`) on a copy of scr4k32, standard vectors | `--batch-log2 24 --warps 2048` | standalone 3/3, in batch 2/3 (base 1,000,000, warp 530's 16th unit, caught), overall FAIL |
| broken tag, 2 units on 1 warp, standard vectors | `--batch-log2 6 --warps 1` | 3/3, 1/1, PASS: missed, the standard vectors have no base 32 |
| 14 edge packs, run A | `--batch-log2 6 --warps 1 --batches 1` | 14/14 PASS, 4/4 standalone and 2/2 in batch each (bases 0 and 32 on one arena) |
| 14 edge packs, run B | `--batch-log2 9 --warps 1 --batch-base 4294967040` | 14/14 PASS, 4/4 and 3/3 each (16 units on one arena, the wrap inside) |
| broken tag on edge-slot0 at 32 and 128 KiB | `--batch-log2 6 --warps 1` | 4/4 standalone, 1/2 in batch, FAIL at both (the second unit read the first's slot 0) |
| broken lazy fill (`m_` all ones) on edge-slot0 at 32 KiB | same | 0/4, 0/2, FAIL |
| 200 fuzz packs | `--batch-log2 9 --warps 2 --batch-base 4294967040 --batches 1` each | 200/200 PASS, 800/800 standalone units (25,600 hashes), 400/400 in the window; 91 s wall for the 200 runs (20:16:02 to 20:17:33 UTC) |
Totals: 228 of 228 PASS where expected, 3 of 3 FAIL where built in. Reading: the one-warp CPU simulation is exact on Metal under the hosts' present tag policy; the analysis names the host contract (zero the arena at allocation and at the tag counter's wrap, tags from 1, groups a multiple of warps) that turns that into a guarantee, and finds layer 3 does not move the named chip (section 3.4 of the write-up: 2.4x at every share under the cap).
## 5 October 2026 (night), epoch length as an era parameter (Counter ASIC 2.0, layer 9): the Mac's compile-ahead per program
Branch `ca2-epoch`, worker "ca2-epoch"; design and the per-card table in `docs/plans/epoch-length.md`. Question (the project lead: "what about faster program changes?"): what a card spends per epoch between receiving the next seed and swapping, which sets the floor of the epoch-length ladder (600 to 7,200 DAA s). Machine: Apple M5 Max (Darwin 25.6.0, 64 GiB), 21:18 UTC, load average 11 to 14 from other agents' builds and runs; the measure lock held for the 3-s run (`tools/lock/with-lock.sh measure bash scratchpad/epoch-measure.sh`). `proto-metal/igneum-bench` built from this branch with `swiftc -O -target arm64-apple-macos11 -o igneum-bench main.swift -framework Metal` under a build slot.
Ten distinct programs (seed strings `igneum-devnet-v4-epoch0`, `/epoch1` .. `/epoch9`; version 2 generator, 128 loads per hash), each generated and compiled at run time (`makeLibrary` from source plus `makeComputePipelineState`), dataset 2^28 words, one 2^20 batch and one verify warp per program:
./igneum-bench --seed igneum-devnet-v4-epoch0 --hours 10 --dataset-log2 28 --batch-log2 20 --batches 1 --verify-warps 1
| Program | Compile ms (library + pipeline) | Mhash/s (GPU) | Verify |
|---|---|---|---|
| epoch0 | 18.8 (9.3 + 9.5) | 27.2 | PASS |
| epoch1 | 17.8 (8.8 + 9.1) | 27.9 | PASS |
| epoch2 | 17.6 (8.6 + 9.1) | 27.8 | PASS |
| epoch3 | 16.1 (8.0 + 8.2) | 27.7 | PASS |
| epoch4 | 18.6 (9.0 + 9.6) | 27.5 | PASS |
| epoch5 | 15.9 (7.7 + 8.2) | 28.4 | PASS |
| epoch6 | 17.6 (8.6 + 9.1) | 27.8 | PASS |
| epoch7 | 17.6 (8.7 + 9.0) | 30.1 | PASS |
| epoch8 | 20.4 (9.8 + 10.6) | 28.4 | PASS |
| epoch9 | 18.2 (8.7 + 9.5) | 29.3 | PASS |
| min / median / max | 15.9 / 17.7 / 20.4 | | 10 of 10 |
Cache fill 1.95 ms GPU (192.4 ms one core), dataset build 20.8 ms GPU for 1 GiB. The devnet pack three times through `packbench --pack ../proto-cuda/packs/igneum-devnet-v4-epoch0 --batches 1 --batch-log2 20 --group 256` (the pack's two libraries, `memhard.metal` and `program.metal`): compile 79 ms, 1 ms, 1 ms (the system shader cache answers the identical source from the second run); cache fill 0.6 to 0.7 ms GPU, dataset build 20.7 to 20.8 ms GPU.
Reading: a fresh program compiles in about 18 ms on this card with the Metal compiler service warm, 79 ms for a pack with its dataset kernels, up to 1.8 s cold (the variant-racing entry's first seed), 0 to 444 ms at the fleet's live boundaries (M11). The hot table fill of layer 5 is 0.07 to 0.22 ms (ca2-cache). So the Mac's per-epoch compile-ahead is under 2 s without the race and about 38 s with it (M11: 34.0 / 34.9 / 37.8 s), and the race is the only item visible against the 600-s window in which the program is known (lead 1,200 s minus the 600-s VDF, fixed at every epoch length). PC cards, cited in the plan: RTX 5090 NVRTC 151 to 180 ms, prepare 0.5 to 1.0 s without the dataset (M11), race one round about 37 s; RX 9070 XT OpenCL compile NOT MEASURED at the current worker (owed: `host.c` times `clBuildProgram` only in the `prepare` path and no `prepared` line from gfx1201 is in any upload); Intel UHD build 3.0 to 6.4 s (M11). Floor by the rule (slowest compile-ahead under 10% of the epoch and inside the window, dataset excluded): 600 DAA s, carried by the race at 6.3% of 600 s; with the race off (M11 found base wins on both the 5090 and the Mac) the slowest measured row is the Intel iGPU at 1.1%. Consequences per tier and the difficulty-settle constraint (24% of a 600-s epoch in settle at the measured 144 s) are in the plan.
### 6 October 2026, 00:4xZ, the empty `/api/state` reply (proving v1 branch)
Reported by the aggregation-cost agent: PC 2's `/api/state` answered `{}` (2 bytes) at 22:22Z, 22:41Z and 00:18Z. Not measured on PC 2 (no job); derived from the app source and node 1's RPC, read-only on the Mac:
| Figure | Value | Source |
|---|---|---|
| Paid shards, devnet, all provers | 663 | `curl -s 127.0.0.1:26790 -d '{"jsonrpc":"2.0","id":1,"method":"igneum_getProvingStatus","params":[]}'` at tip DAA 0x22caf |
| Paid wei, all provers | 0x2c2961a69990745400 = 814.64 IGN | same call |
| Average per paid shard | 1.23 IGN (approximate: the mean over 663) | 814.64 / 663 |
| u64::MAX in IGN | 18.45 | 2^64 - 1 over 1e18 |
| Paid shards per app start before the reply empties | 15 (approximate: at the mean payout) | 18.45 / 1.23 |
Cause: `ProvingState.paid_wei: u128` and serde_json `to_value` (1.0.151, `value/ser.rs` `serialize_u128`: u64 range or an error); the error became `json!({})`. Fix: the field serialises as a decimal string; `state_json` logs the error once. Test `a_paid_total_over_u64_max_still_serialises_the_whole_state` (`cargo test --offline -q paid_wei`, 1 passed).
| The fast-time 3-node harness (`tools/proving-v1/net.mjs`, 29950+, suffix 956, every node in trust mode, three vmine voters, v0 at DAA 60, v1 at DAA 120, 4 blocks a segment, unproven after 60 DAA, a tenth to the aggregator; fork b177718e built on this Mac) | run 2, 19:13:01Z to 19:16:19Z, under the run lock: PASSED, 21 checks in 197.3 s (`tools/proving-v1/report-2026-10-05.json`). v1 start = chain block 119 on all three nodes; the native statement identical on all three. Known-finished: segment 119..122's fresh-chain record submitted to n1 at t=131.1 s, relayed, verified (trust) and PAID on n0 1.0 s later at chain block 129, 253,611,648,000,000,000 wei = a tenth of the four credits, the same on every node, the payout address holding it. Chain rule: segment 123..126's fresh-chain record refused ("does not chain to segment 119..122 ... proven (record paid at chain block 129)"), the continuing one (chain_len 8) accepted and paid. Known-failed: segment 127..130 left without a record: a fresh-chain record for 131..134 refused while 127..130 was pending ("pending until DAA 191"); at DAA 192 the status read unproven, a late record for 127..130 refused ("unproven: carried after the deadline"), the fresh-chain record for 131..134 accepted and paid with chain_len 4; `segmentsInWindow` proven 3, unproven 1. The shard side: a v1 shard's `shardWei` = 90% of its block's credit. Run 1 (19:10Z) failed in its own tooling (the signer's argument order), fixed |
## 5 October 2026 (night), aggregation cost on the RTX 5090: what a per-block aggregation spends and what each lever gives (proving engineer, agg-cost)
the project lead, 5 October 2026: "fix everything else in the numbers tonight". The number under test: the chained segment aggregation cost 9.6 to 9.7 s a block on PC 2's 5090 while the card mined (`chain-pc2-pv1c`, the entry above), 2.2 s on 4 October with the card to itself. Target: under 3 s a block, the miner's slowdown of the prover under 1.5x, the proof statement unchanged. Branch `agg-cost` (worktree `igneum-wt-agg-cost`, from `proving-v1` 219517f). Host changes (statement untouched, `elf/` untouched): the aggregation's stdin build timed apart from the prove call, the deferred-proof count and the SP1 knobs in the RESULT lines, `--mode chain --save-shards` (every shard's compressed proof written next to the results, so `--mode aggregate` re-runs the same proofs under other settings). Jobs: `agg-cost-pc2-1` (21:01:20Z to 21:25:11Z, `tools/proving-v1/pc2-agg-cost.ps1`, the package `igneum-prove-wsl2-aggcost.zip` fetched by `fetch-prove-aggcost` 20:55:39Z, built in WSL2 against the live target dir in 5 s, installed to `/opt/igneum-aggcost`, the live `/opt/igneum` untouched, `--mode id` the pinned pair) and `agg-cost-pc2-2` (21:34:00Z, the same script). The live prover was switched OFF for the runs (its sp1-gpu-server would otherwise be shared through `/tmp/sp1-cuda-0.sock` and carry its own environment; `gpu_server_before running=0`) and ON again at the end. Fixtures: four consecutive live blocks cut from PC 2's own node (86165..86168 at tip 86195, one empty shard each, every one MATCHES natively), the same four for every phase of job 1. App 0.3.9 on PC 2 throughout.
Known-finished case of the host changes before the GPU (this Mac, CPU, run lock, 20:41Z to 20:44Z): `--mode chain` over `fixtures/chain/block-81046.json` with `--save-shards` (shard 38.5 s, aggregate 43.4 s, the proof file written), then `--mode aggregate` over that saved shard proof with `SP1_WORKER_VERIFY_INTERMEDIATES=false` (46.6 s, the same statement `0x3dedb8ea...`), `--mode verify-segment` VERIFIED in 0.027 s; known-failed: a wrong statement NOT VERIFIED in 0.027 s. Unit tests: `cargo test --release -p igneum-prove-core -p igneum-prove-host`: core 8 passed, host 9 passed and 1 ignored (build lock, 20:53Z).
### Lever 1, the profile: where a per-block aggregation goes
| What | Measured (job `agg-cost-pc2-1`) |
|---|---|
| The host's own share of an aggregation (the stdin build: the AggInput, the proof clones into the request) | 0.000 s on every block, mining or idle (the `stdin` field of every `RESULT chain block` line): everything is inside the one `prove().compressed()` call to the GPU server |
| The GPU server's log at `RUST_LOG=info` (phase A0, the same chain of 1, stderr captured) | 1 line: sp1-gpu-server 6.8.1 prints no spans and no timings, so the step costs below are read from the deferred-proof count, not from a profiler |
| Aggregation with 1 deferred proof (the first block, no previous proof) against 2 (every chained block), the card mining | 7.9 s against 9.6, 9.6, 9.8 s: the second deferred proof costs 1.7 to 1.9 s under the miner |
| The same, the miners paused (phase C, the same fixtures, 21:05:51Z) | 1.7 s against 2.1, 2.1, 2.2 s: the second deferred proof costs 0.4 to 0.5 s alone |
| The shard proof of an empty shard | 7.4 to 7.8 s mining, 1.9 to 2.2 s alone |
| A whole block (one empty shard plus its aggregation) | 17.1 to 17.3 s mining (end to end 67.4 s for 4 blocks), 4.1 s alone (16.4 s for 4) |
| GPU utilisation over the phase (1-s `nvidia-smi` samples) | 93.9% mining (80 samples, the miner's), 15.8% alone (32 samples): the prover alone keeps the card busy a sixth of the time. Its work is short GPU bursts between CPU phases (the executor, the witness and recursion-program generation run on the CPU inside the server), and the miner's kernels fill the gaps |
| GPU memory peak | 16,195 MiB mining (the miner's 3.4 GB resident), 14,483 MiB alone |
| The slowdown by the miner, same fixtures, same host, 2 min apart | shards 3.6x, the first aggregation 4.6x, a chained aggregation 4.5x, a block 4.2x |
| Setup per host process (client plus two key setups) | 13.0 to 15.7 s, mining or not |
Reading. An aggregation is three or four recursion steps on the card (the aggregator guest's one core shard, its lift, one deferred program per verified proof, the compose), each a burst of under half a second when the card is free. The chained aggregation's extra deferred proof is the only part that grows with the chain rule, 0.4 to 0.5 s alone. Everything else the 9.7 s holds is the miner: with the card at 94% from the lottery kernels, every prover burst waits for a time slice, and a 2.1-s aggregation becomes 9.7 s. The 4 October 2.2 s (two shards, no previous proof, the card to itself) and tonight's 1.7 s (one shard) and 2.1 s (one shard plus the previous proof) agree within the deferred count.
### Lever 2, batch and tree folds (estimate from the measured step costs; the statement is pinned, no guest was changed tonight)
A fold of K blocks' shard proofs plus the previous segment proof in ONE aggregator call would cost one core shard, one lift, K + 1 deferred programs and the compose tree in place of K chained aggregations. From the measured rows (alone: a 1-deferred aggregation 1.7 s, each further deferred proof 0.45 s; mining: 7.9 s and 1.8 s):
| Fold | Deferred proofs per call | Per block, card alone (estimate) | Per block, card mining (estimate) | Rule |
|---|---|---|---|---|
| chained, as pinned (measured) | 2 | 2.1 s | 9.7 s | one call per block |
| batch of 4 | 5 | (1.7 + 4 x 0.45) / 4 = 0.9 s | (7.9 + 4 x 1.8) / 4 = 3.8 s | one call per 4 blocks |
| batch of 8 | 9 | (1.7 + 8 x 0.45) / 8 = 0.7 s | (7.9 + 8 x 1.8) / 8 = 2.8 s | one call per 8 blocks |
| tree of 4 (2 + 2, then the pair) | 3 per call, 3 calls | 3 x (1.7 + 2 x 0.45) / 4 = 1.9 s | 3 x (7.9 + 2 x 1.8) / 4 = 8.6 s | no gain over the chain: every call pays the fixed part |
Reading. A batch fold halves to quarters the per-block aggregation but changes the aggregator's statement (`AggInput` carries one block's shards and the guest asserts one block hash), so it is a new pinned guest and a new program id: a provers-off drain and a rollout (proving/README.md, pinned guests). It does not reach 3 s on a mining card by itself (2.8 s at K = 8 is on the line), and the shard proof beside it stays 7.4 s a block on a mining card. The lever that moves both is the card's other job, lever 4. A tree fold gains nothing here because the fixed part of a call (the core shard and the lift) dominates the per-proof part 4 to 1.
### Levers 3 and 4, two streams and the miner's kernels (job `agg-cost-pc2-2` and the re-run)
Job `agg-cost-pc2-2` (21:34:00Z to 21:49:22Z) ran with the 5090 idle throughout: job 1's `/api/resume` had left the worker off (below), so the rows that needed the miner (the batch-log2 curve, the two streams beside the miner, the time-slice policy, the chosen combination) are void and wait for a re-run; the idle rows are measured.
| What | Measured (job `agg-cost-pc2-2`, card idle) |
|---|---|
| Aggregate-only over job 1's four saved shard proofs (`--mode aggregate --proofs b1;b2;b3;b4 --parent ...`, one process, the same statement `0x3a995f24...` as the chain run), default knobs (phase B0, then C1) | 1.7, 2.0, 2.0, 2.0 s (1, 2, 2, 2 deferred proofs), 8.1 s for four; C1: 1.8, 2.1, 2.1, 2.1 s, 8.3 s |
| The same with `SP1_WORKER_VERIFY_INTERMEDIATES=false` (phase B; the server inherits the host's environment, the knob printed in the `sp1 knobs` line) | 1.7, 2.0, 2.0, 2.0 s, 7.8 s for four: no gain (0.3 s over four, inside the run-to-run spread of 0.2 s). The knobs that change the recursion shape (`SP1_WORKER_MAX_COMPOSE_ARITY`, `MAX_REDUCE_ARITY`) were not tried: a different shape is a different recursion key set and the pinned verifier would refuse the proof |
| A 4-deferred aggregation (block-344-shards4, four prototype shards of 6.75 M pgas, phase C2) | shards 42.8 s (10.7 s each, the 4 October 10.2 to 10.7 s), aggregation 2.4 s with 4 deferred proofs; GPU peak 28,402 MiB (the prototype shard's 28.3 GB), utilisation 27.7% over the phase. With 1.7 s at one deferred proof and 2.0 to 2.1 s at two: 0.25 s per further deferred proof alone, so a batch of 8 would cost about 3.5 s a call, 0.45 s a block (estimate, the pinned statement forbids it) |
| Two host processes at once on the one card (phase G0: chains of 2 on disjoint blocks, started 2 s apart) | both connected to ONE sp1-gpu-server (the first process's child; the socket is per device, `/tmp/sp1-cuda-0.sock`): process 1 shard 2.2 and 3.5 s, aggregation 3.0 and 4.0 s (12.9 s for 2 blocks against 8.2 s alone); process 2 shard 3.3 s, aggregation 3.6 s, then its second block died with `CudaClientError: Failed to read the response: early eof` when process 1 finished and its server exited. GPU 24,911 MiB, utilisation 12.4% and 13.1%. Two streams through SP1 6.8.1's server are serialised on one socket and the second dies with the first: no throughput gain (3 blocks in 33 s against 4 in 16.4 s) and a failure mode; lever 3 is closed on this SP1 version |
| Job 3 (`agg-cost-pc2-3`, 22:41:15Z, app 0.3.10, the same script with the socket rule and a card switch): phase A, the app's 5090 miner at 117.0 MH/s mean (n 3, STATUS lines 22:44:45Z to 22:46:11Z), four fresh live blocks 90896..90899 | shards 8.0, 7.8, 7.6, 7.8 s; aggregations 8.0 s (1 deferred), 10.0, 10.0, 10.0 s (2 deferred); 69.5 s for four, 17.8 s a block; GPU 93.8%, peak 16,245 MiB: the job-1 baseline reproduced 100 min later on other blocks |
| Job 3's own-miner phases | void: the state reads came back empty (the class below), the card switch did nothing, phase D launched my miner beside the app's (the app's dropped to 62.2 MH/s, mine read 60.6 MH/s), then PC 2's app restarted at 23:03:30Z and the job died with it; no curve point |
| The GPU time-slice policy (`nvidia-smi compute-policy --set-timeslice`, the restore job `agg-cost-restore-1`, 23:16:53Z) | "Not Supported" on PC 2 (RTX 5090, driver 13.3, the Windows nvidia-smi, not elevated): the lever is closed on this driver; an elevated try is not worth a slot, the error is the driver's, not a permission's |
| The own-miner phases of job 2 | void: no 5090 miner was running to copy the command line from (the worker off since 21:25Z) |
The curve, job `agg-cost-pc2-6` (01:12:09Z to 01:24:14Z, app 0.3.11, PC 2 to itself; every phase closed before the next job landed on PC 2 at 01:24:21Z). The app's 5090 miner switched off through `/api/cards` (the keys from `settings.json`; the worker was still alive after 120 s, `/api/pause` as the fallback stopped it in 5 s), then the job's OWN miner on the 5090 with the app's command line (`igneum-miner mine ... --worker igneum-worker-cuda.exe --identities 8 --worker-args "--device 0 --pack packs\devnet --race off [--batch-log2 B]"`, the base variant, its STATUS line every 10 s), the same four live blocks 96556..96559 (one empty shard each) proven by `--mode chain` under it, the miner's rate from its own `now=` field (the first two lines skipped). `--batch-log2 B` sets the worker's nonces per kernel launch (2^B; 22 is the worker's default, 4,194,304 nonces, about 35 ms a launch at 120 MH/s; `proto-cuda/nvrtc/worker.cpp`).
| batch-log2 | Shard proof (4, s) | Aggregation (1 deferred, then 2) (s) | A block (s) | GPU util. (%) | GPU peak (MiB) | Own miner (MH/s wall, n) | Against the card alone (4.1 s a block) |
|---|---|---|---|---|---|---|---|
| 22 (the default), phase D | 8.1, 7.8, 7.9, 7.8 | 8.4; 10.3, 10.0, 10.4 | 18.1 | 95.5 | 16,580 | 103.9 (9) | 4.4x |
| 20, E20 | 8.1, 7.8, 7.8, 7.8 | 8.4; 10.3, 10.1, 10.1 | 18.0 | 94.9 | 16,461 | 103.7 (8) | 4.4x |
| 18, E18 | 7.0, 6.7, 6.7, 6.7 | 7.2; 8.9, 8.8, 8.8 | 15.6 | 91.5 | 16,487 | 99.3 (7), minus 4.4% | 3.8x |
| 16, E16 | 5.1, 4.9, 4.9, 4.9 | 5.0; 6.1, 6.2, 6.2 | 11.1 | 85.3 | 16,519 | 83.8 (6), minus 19% | 2.7x |
| 16 again, phase H (the job's own choice: the shortest chain) | 5.0, 4.9, 4.8, 4.9 | 4.9; 6.1, 6.2, 6.2 | 11.1 | 85.7 | 16,487 | 84.0 (6) | 2.7x |
Reading. Between 2^22 and 2^20 nothing moves: the card's time-slice scheduler alternates the two contexts whatever the kernel length above a few milliseconds. From 2^18 down the miner's launches get short enough (about 2 ms at 2^18, 0.5 ms at 2^16) that the prover's bursts find the card sooner, and the miner pays in launch overhead and idle gaps: at 2^16 the prover runs 1.6x faster (18.1 to 11.1 s a block, the chained aggregation 10.2 to 6.2 s) for a fifth of the hash rate, and it is still 2.7x slower than on a card to itself. The trade is about 1 MH/s per 0.37 s of block time at the 2^16 point, and the 3-s aggregation and the 1.5x slowdown are not reachable on a mining card by the kernel length; a 2^14 point (approximate, extrapolated) would be about 8 s a block at about 65 MH/s. The phase E0 (a 4-deferred aggregation under the miner) failed in 0.1 s: its proof paths pointed at `/` where job 1 had left its shard proofs, but job 2's block-344 proofs sit in job 2's own folder (`$JOB` was exported from job 2 on); the 4-deferred cost under the miner stays an estimate (lever 2 above). The app's own 5090 miner ran at 117 MH/s (job 3, 22:44Z) and 110 to 129 MH/s (its STATUS lines at 01:10Z) with the prover beside it, against my miner's 104 MH/s at the default batch: my miner runs the base variant with `--race off` (no tuning file on PC 2), so the curve's rates are relative to each other, not to the app's.
### Lever 5, the host side under WSL2 (what the chain-mode numbers leave out)
| What | Measured |
|---|---|
| The export (`igneum_exportSegments` 0..tip, 75 to 77 MB over curl.exe to a file on `C:`) | 1.1 to 1.5 s |
| The cut (`igneum-prove-export` replaying from genesis, then `--mode native`), four blocks | 18 s for four including the native checks (21:01:28Z to 21:01:46Z), about 4 s a block; the export's file sits on `/mnt/c` |
| The key setup per host process | 13.0 to 15.7 s on PC 2 (8.0 to 8.5 s on the Mac CPU): `--mode chain` and `--mode aggregate` pay it once per process, the app's loop pays it per shard |
| The proof file write through the WSL2 bridge | the 4 October entry ("shard proving on the RTX 5090"): 24 min of unbuffered `save` across `/mnt/c`, fixed by the 4 MB buffer; tonight `--save-shards` wrote the four 1.27 MB proofs inside the chain phase with no visible gap (the A phase's 80.4 s wall against 67.4 s of proving plus 13.0 s of setup) |
| Native Linux | not measured: no native Linux machine with an NVIDIA card exists in the project tonight, and the 4 October numbers were also WSL2 (Ubuntu 24.04 under PC 2's Windows). The WSL2 cost inside a `prove()` call is not separable from here; the host-side pieces above are what a native box would also skip or keep |
### What went wrong, measured
| What | Fixed |
|---|---|
| Job 1's per-phase command ran with `$JOB` empty (the bash variables of `vars.sh` were set, not exported, and the command runs in a child bash): `--out /results-A.json`, the saved shard proofs in `/` on the WSL root, so the aggregate-only phases B0, B, C1 and the prototype-shard phase C2 failed in 0.0 s ("No such file") | `export` in `vars.sh`; job 2 reads the proofs from `/` |
| Job 1's own-miner phases launched the iGPU miner (the first `igneum-miner mine` process matched; the 5090's is the second) and `if (StartMiner ...)` was always true (PowerShell: a function's emitted RESULT strings are part of its output), so D and E ran with the 5090 idle and the AMD iGPU at 3.4 MH/s: three more idle replicates of the chain (2.0 to 2.2 s shards, 1.8 and 2.2 s aggregations), no curve | the miner matched on `igneum-worker-cuda`, the outcome in a script-scope flag, `--race off` for the own miner (no tuning file on PC 2; a race costs up to 120 s a start) |
| Job 1's `/api/resume` at 21:25:11Z answered ok and the 5090 miner stayed off (card state `off`, hash 0.0, 1,760 MiB on the card) until the 0.3.10 restart; job 2 waited its full 600 s for a hash rate and ran its mining phases void | the restore job `tools/proving-v1/pc2-agg-cost-restore.ps1` also posts `/api/start`; the Counter ASIC coordinator opened a task chip for the resume defect |
| Jobs 3 and 4 (`agg-cost-pc2-3` 22:41Z on app 0.3.10, `agg-cost-pc2-4` 00:18Z on 0.3.11): every `/api/state` read came back as the two bytes `{}` (job 4's raw-body print: `raw_len=2`; the same reads gave the full state on 0.3.9 at 21:01Z and the AMD agent saw the empty reply at 22:22Z), so the card switch found no card, the app's 5090 miner kept mining, and job 3 ran a second miner beside it (two miners at about 60 MH/s each) while job 4's double-mining guard voided its own-miner phases. The class is the app's, not the reader's: `state_json()` (engine.rs:180) does `serde_json::to_value(st).unwrap_or(json!({}))`, and the value that fails is `ProvingState.paid_wei: u128` (serde_json 1.0.151 refuses a u128 over u64::MAX, 18.45 IGN; the proving-v1 agent's diagnosis): a paid shard averages 1.23 IGN, so the reply empties about 15 paid shards after every app start and comes back at the next restart, which matches the times (full at 21:01Z with paid_wei 0, empty from 22:22Z after the prover had paid from 22:02Z). Fixed on the app branch proving-v1 at 6714a45 (paid_wei as a decimal string, the error logged, an `{"error":...}` reply on any future failure) | job 5 reads the card keys from the app's `settings.json` (`cards`: key to enabled and identities), restores the 5090's 8 identities first (the restore job of 23:16:53Z had set 2: its parser read the next card's value), refuses before any pause when it cannot name the card, waits on the CUDA worker process count for the card to stop, and checks the worker is back at the end |
| Job 5 (`agg-cost-pc2-5`, 01:10:44Z) failed at PowerShell's parse in 1 s: `$RestoreIdentities:` inside a double-quoted string (a drive-qualified variable); no card or miner touched | `${RestoreIdentities}:`; the other `$name:` shapes are inside single-quoted bash here-strings |
| Job 6's identities step found `settings.json` already at 8 identities under the active key `nvidia:0:NVIDIA GeForce RTX 5090` (a stale key `nvidia:NVIDIA GeForce RTX 5090` carries 2), so no change was sent; job 6's `/api/cards` with the 5090 disabled answered ok but the worker ran on for 120 s, `/api/pause` stopped it in 5 s, and at the end `/api/resume` brought it back in 5 s on 0.3.11 | the card switch keeps the pause as its fallback; the resume path works on 0.3.11 |
| PC 2 ran three jobs at once from 01:24Z (`run-prover-on-pc2-20261006` at 01:24:21Z, the ledger suites build at 01:26:15Z, while agg-cost-pc2-6's closing report was still being uploaded): the app does not serialise jobs, "one job per machine at a time" holds only by the coordinator's word; job 6 had closed at 01:24:14Z, so its rows are clean | nothing of mine to fix; a rule for the job runner |
| The make-package gate ran the exporter's side files (`block-N.json.node-plan.json`) as fixtures and failed; its execute step took the exclusive `measure` lock for a cycle count and queued 25 min behind a packbench run | the glob skips `.node-plan.json`; the execute step runs under the `run` lock (a count, not a time) |
### 6 October 2026, 07:12Z to 07:17Z, the host's chain mode with --save-shards records and --prev, on the Mac's CPU
`tools/lock/with-lock.sh run`, `SP1_PROVER=cpu igneum-prove-host --mode chain --chain proving/fixtures/chain/block-81046.json,block-81047.json --save-shards --out chain-a.json`, then `--chain block-81048.json --save-shards --prev segment-81047-aggregated.bin --out chain-b.json` (the app branch at ce8f34a, Apple M5 Max, CPU prover). The flags the app's segment path needs, before PC 2 (approximate figures: a CPU run, one sample each):
| Step | Value |
|---|---|
| Shard proof, CPU, empty block | 34.7 s and 36.3 s |
| Aggregation, CPU, 1 then 2 deferred proofs | 39.1 s, 50.8 s |
| Chain of 2, end to end | 160.9 s |
| Per-shard records written | 2 (number, block_hash, shard, statement, proof_sha256, proof_bytes 1,272,897, proof_file, prove_seconds) |
| `--prev` run: base_chain_len, final chain_len | 2, 3 (the chain continued; a wrong previous proof is refused by number and parent hash) |
### 6 October 2026, 07:52Z to 08:24Z, the segment-aligned prover beside the miner on PC 2's RTX 5090 (job `segments-pc2-pv1c`)
`tools/proving-v1/pc2-segments.ps1` (app branch 330207d; the host from the package `igneum-prove-wsl2-segal`, built on PC 2 in 7 s warm to `/opt/igneum-segal`, pinned guests unchanged); the app's own prover OFF for the run through `/api/prove`, ON again at the end; the app's miner running (8 identities, batch-log2 22); `SP1_PROVER=cuda`, the stock 6.8.1 GPU server; a 1-s nvidia-smi sampler under every chain. Payouts read on node 1 (read-only, `igneum_getProofRecords` per block at 08:30Z). The miner's rate from the app's uploaded log (`status: ... MH/s` every 30 s, run win-1ccfe586-20261005-235130).
| Figure | Value | Note |
|---|---|---|
| Segments claimed in 30 min | 9 (114470, 114654, 114862, 115022, 115198, 115366, 115542, 115710, 115870) | one every 210 s; 32.1 min of loop |
| Candidates per pass | 32 to 38 whole segments inside the margin | margin 580 to 589 DAA at claim |
| Export (the chain to the segment's last block) | 99.6 to 100.6 MB in 1.4 to 1.6 s | once per segment |
| Cut (8 fixtures, the exporter) | 45.1 to 46.0 s | the exporter replays from genesis per block; the next lever |
| Chain run wall (8 shards, 8 aggregations, one key setup) | 159.7 to 160.6 s | host `--mode chain --save-shards` |
| Shard proofs, 8 per segment | 63.0 to 63.5 s (7.9 s a shard) | empty blocks |
| Aggregation, 8 chained | 80.4 to 81.1 s (10.1 s a block) | the fixed cost per block beside the miner |
| End to end per segment (export, cut, chain, sign, submit) | 210.0 to 211.2 s | |
| GPU memory peak during a chain | 16,484 to 17,573 MiB (miner resident) | the 24 GB tier's gate holds |
| GPU utilisation during a chain | 94.9 to 95.3% | |
| Shard records accepted | 72 of 72 | 8 per segment |
| Shard records paid on chain | 72 of 72 | 0.905 to 2.719 IGN a shard (90% of the credit); carried 180 to 226 blocks after the block |
| Segment records accepted | 0 of 9 | every one refused: "does not chain to segment N-8..N-1 (chain_len 8), which is pending until DAA ..." |
| Miner alone (the app's prover off), 07:25 to 07:51Z | 117.86 MH/s mean (n=52) | min 46.37 is the switch-off dip at 07:22Z |
| Miner beside the segment prover, 07:55 to 08:24Z | 104.90 MH/s mean (n=58, min 98.39, max 119.24) | 12.96 MH/s = 11.0% of the miner, at 95% GPU utilisation from the prover |
| The 0.3.11 prover as shipped beside the miner (5 October row) | 5.0 MH/s = 4.0% | one shard per 46 s; this run proves 8 shards per 210 s, 2.8x the shards |
| Node 1's v1 window at 08:24Z | pending 59, proven 0, unproven 16, paid segments 0 | unchanged by the run: the chain rule |
What the refusal is (the fork, `igneum/exec/src/proving.rs` `check_segment_record`): a fresh record (chain_len = N) is valid only when the previous segment is UNPROVEN at the carrier, and the record's own deadline is the previous segment's deadline plus one segment length in DAA, so a fresh record is valid for 8 DAA (about 8 s) per segment and must be carried inside them. With one prover every previous segment is pending at proof time. Fixed on the fork branch behind `proving_v1_fresh_rule_daa` (0f0dda95): from the switch a fresh record is valid whenever the previous segment is not proven; the app holds a refused record and offers it again every pass until the deadline (272b025).
Run b (`segments-pc2-pv1b`, 07:20Z to 07:51Z) claimed nothing in 88 passes: the driver's segment keys were doubles against int64 hashtable keys (fixed in 330207d); its 30 minutes are the miner-alone baseline above.
### 6 October 2026, 08:26Z to 08:35Z, the fast-time harness on the fresh-record rule (Mac, `tools/lock/with-lock.sh run`)
`IGNEUM_PV1_BIN=vendor/igneum-node/target-pv1/release node tools/proving-v1/net.mjs --segment 8 --unproven 10 [--fresh-rule 0]` (fork 0f0dda95, 3 nodes at 60x, ports 29950+):
| Case | Checks | Time |
|---|---|---|
| The rule as shipped (no switch): fresh refused while the previous segment is pending (known-failed), accepted after it is unproven | 22 passed | 166.2 s |
| `--fresh-rule 0`: fresh accepted while the previous segment is pending, `freshAdmissible` true, still refused after a proven one, the second offer a duplicate ("segment already paid") | 23 passed | 139.9 s |
## 5 to 6 October 2026, prover floor: the SP1 6.8.1 GPU server rebuilt for small cards (prover-floor agent)
Branch `prover-floor` (worktree `igneum-wt-prover-floor`); the model with file and line, the patch and the reading

View file

@ -0,0 +1,301 @@
{
"tool": "tools/txgen/relay-net.mjs",
"node": "/Users/joshm/Projects/igneum/vendor/igneum-node/target-txgossip/release/igneumd",
"topology": "A - B - C (B dials A and C); generator on A; vmine on B and C",
"params": {
"rate_per_s": 2,
"duration_s": 120,
"wallets": 16,
"fund_ign": 2,
"fast_time": true
},
"chain_blocks_in_window": 149,
"executed": 256,
"skipped": 0,
"by_miner": {
"B": {
"blocks": 79,
"blocks_with_txs": 59,
"txs_carried": 151,
"executed": 151,
"skipped": 0
},
"C": {
"blocks": 70,
"blocks_with_txs": 42,
"txs_carried": 105,
"executed": 105,
"skipped": 0
}
},
"generator": {
"sent": 240,
"included": 240,
"pending_at_end": 0,
"included_per_s": 1.975,
"latency_ms": {
"p50": 1545,
"p90": 3058,
"p99": 5033,
"max": 6017,
"mean": 1859
},
"blocks_with_content": 100,
"errors": {},
"stop_reason": "duration"
},
"pools": {
"samples": [
{
"t": 33.5,
"A": 0,
"A_chain": 37,
"B": 0,
"B_chain": 37,
"C": 0,
"C_chain": 37
},
{
"t": 38.5,
"A": 6,
"A_chain": 38,
"B": 6,
"B_chain": 38,
"C": 6,
"C_chain": 38
},
{
"t": 43.5,
"A": 2,
"A_chain": 44,
"B": 2,
"B_chain": 44,
"C": 2,
"C_chain": 44
},
{
"t": 48.5,
"A": 2,
"A_chain": 50,
"B": 2,
"B_chain": 50,
"C": 2,
"C_chain": 50
},
{
"t": 53.5,
"A": 1,
"A_chain": 58,
"B": 1,
"B_chain": 58,
"C": 1,
"C_chain": 58
},
{
"t": 58.5,
"A": 2,
"A_chain": 66,
"B": 2,
"B_chain": 66,
"C": 2,
"C_chain": 66
},
{
"t": 63.6,
"A": 4,
"A_chain": 71,
"B": 4,
"B_chain": 71,
"C": 4,
"C_chain": 71
},
{
"t": 68.6,
"A": 2,
"A_chain": 77,
"B": 2,
"B_chain": 77,
"C": 2,
"C_chain": 77
},
{
"t": 73.6,
"A": 0,
"A_chain": 80,
"B": 0,
"B_chain": 80,
"C": 0,
"C_chain": 80
},
{
"t": 78.6,
"A": 3,
"A_chain": 83,
"B": 3,
"B_chain": 83,
"C": 3,
"C_chain": 83
},
{
"t": 83.6,
"A": 0,
"A_chain": 92,
"B": 0,
"B_chain": 92,
"C": 0,
"C_chain": 92
},
{
"t": 88.6,
"A": 0,
"A_chain": 96,
"B": 0,
"B_chain": 96,
"C": 0,
"C_chain": 96
},
{
"t": 93.6,
"A": 2,
"A_chain": 101,
"B": 2,
"B_chain": 101,
"C": 2,
"C_chain": 101
},
{
"t": 98.6,
"A": 1,
"A_chain": 114,
"B": 1,
"B_chain": 114,
"C": 1,
"C_chain": 114
},
{
"t": 103.6,
"A": 5,
"A_chain": 117,
"B": 5,
"B_chain": 117,
"C": 5,
"C_chain": 117
},
{
"t": 108.6,
"A": 4,
"A_chain": 121,
"B": 4,
"B_chain": 121,
"C": 4,
"C_chain": 121
},
{
"t": 113.6,
"A": 2,
"A_chain": 125,
"B": 2,
"B_chain": 125,
"C": 2,
"C_chain": 125
},
{
"t": 118.6,
"A": 1,
"A_chain": 128,
"B": 1,
"B_chain": 128,
"C": 1,
"C_chain": 128
},
{
"t": 123.6,
"A": 0,
"A_chain": 135,
"B": 0,
"B_chain": 135,
"C": 0,
"C_chain": 135
},
{
"t": 128.6,
"A": 3,
"A_chain": 140,
"B": 3,
"B_chain": 140,
"C": 3,
"C_chain": 140
},
{
"t": 133.6,
"A": 0,
"A_chain": 144,
"B": 0,
"B_chain": 144,
"C": 0,
"C_chain": 144
},
{
"t": 138.6,
"A": 1,
"A_chain": 155,
"B": 1,
"B_chain": 155,
"C": 1,
"C_chain": 155
},
{
"t": 143.6,
"A": 3,
"A_chain": 163,
"B": 3,
"B_chain": 163,
"C": 3,
"C_chain": 163
},
{
"t": 148.6,
"A": 2,
"A_chain": 167,
"B": 2,
"B_chain": 167,
"C": 2,
"C_chain": 167
},
{
"t": 153.6,
"A": 1,
"A_chain": 175,
"B": 1,
"B_chain": 175,
"C": 1,
"C_chain": 175
},
{
"t": 158.6,
"A": 0,
"A_chain": 179,
"B": 0,
"B_chain": 179,
"C": 0,
"C_chain": 179
},
{
"t": 163.6,
"A": 0,
"A_chain": 184,
"B": 0,
"B_chain": 184,
"C": 0,
"C_chain": 184
}
],
"max": {
"A": 6,
"B": 6,
"C": 6
}
},
"sinks_agree": true,
"wall_s": 168.8
}

View file

@ -0,0 +1,269 @@
{
"tool": "tools/txgen/run.mjs",
"rpc": "http://127.0.0.1:29703",
"chain_id": 4463,
"started": "2026-10-05T17:49:40.914Z",
"ended": "2026-10-05T17:51:44.556Z",
"stop_reason": "duration",
"params": {
"wallets": 16,
"rate_per_s": 2,
"duration_s": 120,
"fund_ign": "2.000000",
"cap_ign": "74.000000",
"gas": 59650,
"stale_s": 60
},
"fees_last": {
"base_gwei": "100.00",
"proving_base_gwei": "10000.00",
"tip_gwei": "1.00",
"node_quote_gwei": "243.86",
"fee_cap_gwei": "243.86"
},
"counts": {
"sent": 240,
"included": 240,
"failed": 0,
"dropped": 0,
"skipped": 0,
"reverted": 0,
"nonceRetries": 0,
"deferred": 0,
"throttled": 0,
"fundingTx": 16,
"pending_at_end": 0
},
"throughput": {
"included_per_s": 1.975,
"send_span_s": 121.5,
"sent_per_s_target": 2
},
"latency_ms": {
"p50": 1545,
"p90": 3058,
"p99": 5033,
"max": 6017,
"mean": 1859
},
"blocks": {
"with_content": 100,
"first": 38,
"last": 176,
"max_tx_in_one": 10,
"per_block": {
"38": 8,
"39": 1,
"40": 2,
"42": 3,
"44": 5,
"45": 3,
"46": 1,
"48": 1,
"50": 2,
"51": 1,
"52": 3,
"54": 2,
"57": 3,
"58": 2,
"59": 1,
"61": 2,
"62": 2,
"63": 1,
"65": 1,
"66": 3,
"68": 4,
"70": 1,
"71": 5,
"72": 1,
"73": 2,
"74": 2,
"75": 2,
"77": 2,
"79": 10,
"80": 1,
"81": 4,
"82": 2,
"83": 3,
"84": 2,
"85": 3,
"87": 1,
"88": 2,
"89": 1,
"91": 1,
"92": 4,
"93": 1,
"94": 3,
"95": 2,
"96": 2,
"97": 1,
"98": 5,
"101": 2,
"103": 1,
"104": 1,
"105": 2,
"106": 2,
"109": 1,
"110": 1,
"112": 1,
"114": 3,
"116": 3,
"117": 8,
"119": 2,
"120": 1,
"121": 6,
"122": 1,
"123": 3,
"124": 2,
"125": 6,
"126": 1,
"127": 4,
"128": 3,
"129": 1,
"130": 4,
"132": 1,
"133": 1,
"134": 1,
"135": 2,
"138": 4,
"139": 1,
"140": 4,
"141": 6,
"142": 3,
"146": 1,
"147": 1,
"148": 3,
"151": 2,
"152": 1,
"153": 1,
"155": 4,
"157": 1,
"158": 1,
"160": 1,
"162": 1,
"163": 5,
"164": 1,
"165": 4,
"166": 1,
"167": 5,
"168": 1,
"169": 1,
"172": 3,
"174": 1,
"175": 3,
"176": 2
}
},
"spend_ign": {
"funding": "32.000000",
"fees_actual": "0.542976",
"fees_max_committed": "38.769773",
"value_moved_between_wallets": "0.132776",
"cap": "74.000000",
"cap_hit": false
},
"wallets": [
{
"index": 0,
"sent": 15,
"balance_ign": "1.921973",
"nonce": 15
},
{
"index": 1,
"sent": 15,
"balance_ign": "1.922278",
"nonce": 15
},
{
"index": 2,
"sent": 15,
"balance_ign": "1.923568",
"nonce": 15
},
{
"index": 3,
"sent": 15,
"balance_ign": "1.924152",
"nonce": 15
},
{
"index": 4,
"sent": 15,
"balance_ign": "1.920937",
"nonce": 15
},
{
"index": 5,
"sent": 15,
"balance_ign": "1.922478",
"nonce": 15
},
{
"index": 6,
"sent": 15,
"balance_ign": "1.924611",
"nonce": 15
},
{
"index": 7,
"sent": 15,
"balance_ign": "1.930741",
"nonce": 15
},
{
"index": 8,
"sent": 15,
"balance_ign": "1.923430",
"nonce": 15
},
{
"index": 9,
"sent": 15,
"balance_ign": "1.926285",
"nonce": 15
},
{
"index": 10,
"sent": 15,
"balance_ign": "1.922495",
"nonce": 15
},
{
"index": 11,
"sent": 15,
"balance_ign": "1.918612",
"nonce": 15
},
{
"index": 12,
"sent": 15,
"balance_ign": "1.921628",
"nonce": 15
},
{
"index": 13,
"sent": 15,
"balance_ign": "1.921379",
"nonce": 15
},
{
"index": 14,
"sent": 15,
"balance_ign": "1.923181",
"nonce": 15
},
{
"index": 15,
"sent": 15,
"balance_ign": "1.923212",
"nonce": 15
}
],
"funder": {
"balance_ign": "203.751501"
},
"lost": [],
"pending_at_end": [],
"errors": {}
}

View file

@ -7,6 +7,7 @@ shard run reported as exit 0, 7a7e873).
| Date | Symptom | Cause | Fix | Proven by |
|---|---|---|---|---|
| 4 Oct 2026 | Every `ci` run on master red since 67bf226 (eleven pushes), unnoticed | `sim/difficulty/records/testnet-v2-2026-10-04.schedule.log` carried a home path; `.log` was outside the identity scrub's extension list in `tools/ci/identity-check.sh` (and in the mirror's `tools/sync.sh`) | 2996cca: `.log` scrubbed like the other text files; the record rewritten with `~`; the same list in igneum-public `tools/sync.sh` (local commit e18256d, not pushed) | `bash tools/ci/identity-check.sh` 0 hits locally; run 37226816xxx on master green |
| 5 Oct 2026 | PC 1 (Windows 11 Pro 26200, default terminal Windows Terminal 1.24): "Windows Command Processor" windows whenever a remote job runs (the project lead) | measured, not guessed: `tools/windows/console-watch.ps1` (job run-20261005-182528) started every candidate child from the app's job runner, whose console is headless (`conhost.exe 0x4`, hwnd 0), with a user32 EnumWindows sampler every 30 ms: powershell, cmd, query, curl, nvidia-smi, wsl --status, a distro, interop cmd and powershell, `powershell -WindowStyle Hidden`, `Start-Process -WindowStyle Hidden`: 0 windows each; `Start-Process cmd` in a new console: a Terminal window and a cmd PseudoConsoleWindow (the known-failed case fires). The 25-minute background watcher (console-watch-bg.ps1, run-20261005-184330, 18:44 to 19:09 UTC, every 200 ms) across an app restart, a build job, two run jobs, two collect jobs and the sweep helper's elevated launch at 19:04:43: 0 console or Terminal windows, 69 conhost starts (every one `conhost.exe 0x4`, headless, under curl, wsl, wslhost, powershell), 1 cmd.exe (under wslhost, WSL interop, no window). The one road that creates a console of its own is the elevated launch (`Start-Process -Verb RunAs`, the AppInfo service: the power cap, the sweep helper, the clock sync, an elevated job); it carried `-WindowStyle Hidden` in four copies, and "Windows Command Processor" is also the name on the UAC prompt the engine raises for cmd.exe (the sweep helper prompted at 17:00, 17:30 and 18:12 UTC, the power cap at every start; the elevated watcher's own prompt, run-20261005-184610, timed out unanswered at 122 s) | `platform::elevated_ps_line` + `elevated_command`: one builder for every elevated launch, hidden by construction, exit 251 when the prompt is refused; the elevated job wrapper reports its own console (`elevated console: hwnd N visible False`) on every elevated job; `tools/ci/windows-spawn-check.mjs` fails CI on a Command::new without the quiet flag, a creation_flags other than CREATE_NO_WINDOW, a Start-Process without -WindowStyle Hidden/-NoNewWindow, or a host.cpp spawn without CREATE_NO_WINDOW / SW_HIDE | the watcher's known-failed case (2 windows) and known-finished case (0); the CI check's self-test (9 cases) and the tree (0 hits); the igneum-app test suite on PC 1 |
| 4 Oct 2026 | `collect-pc1-board3` printed PowerShell parse errors (`.Name`, `.AdapterRAM`) | the publishing shell expanded `$_` inside double quotes to nothing before the command reached the jobs file; nothing to do with Format-List or Out-String (board2 and board4 printed their values) | publish-jobs.sh refuses a collect command that pipes into a script block without `$_` or `$PSItem` | the eaten form refused with the reason, the single-quoted form published to a test folder |
| 4 Oct 2026 | the same job reported `done (exit 0)` over `command exit Some(1)` | `run_collect` in `app/igneum-app/src/jobrun.rs` builds `Done` from the upload count only; the command's exit code is logged and dropped | branch `bugfix-collect-exit`, 35ccdc8 rebased on c257444 (app engine; merge by the main session) | `cargo test --bin igneum-app`: all 28 tests pass on the rebased branch; the new one covers the board3 shape (`Some(1)` is failed exit 1), `Some(0)` done, the cap as timeout, failed uploads still failing |
| 4 Oct 2026 | `publish-jobs.sh --deploy` said "not reachable, differs from the local one, or does not verify yet" after a deploy that had succeeded | one check the instant the CLI returned, while the edge still served the previous file; the deploy's own exit status was hidden by `\|\| true` | `verify_live`: up to `--tries` (12) checks 5 s apart, each failure names its condition; `publish-jobs.sh verify` re-checks on its own; a failed deploy stops before the check | finished: `verify --tries 2` against the live file (try 1 of 2); failed: a local server with an older file ("differs", both publish stamps named) and a closed port ("is not reachable") |

View file

@ -79,6 +79,8 @@ Worked example. Sender S has nonce 5. Miner A's block carries S:5, S:6. Miner B'
Consequence for users. A transaction can be skipped in one block and execute in a later one without being re-broadcast, as long as a miner includes it again; the node's mempool re-queues a skipped transaction once (then drops it). `eth_getTransactionReceipt` returns null until the executing copy lands, as on Ethereum for a pending transaction.
Relay (implemented 5 October 2026, fork `tx-gossip`, protocol version 14). A node's mempool is no longer only what its own RPC received. Every admitted hash is announced to every relay-aware peer within 250 ms (`IgneumEvmTxInvMessage`, the inventory pattern of Kaspa's `InvTransactions`); a peer requests the hashes it does not know (`IgneumRequestEvmTxsMessage`) and the holder answers one `IgneumEvmTxsMessage` with the raw bytes it still has; the receiver runs the same admission as `eth_sendRawTransaction` (signature, chain id, nonce window of 16, fee cap at or above the execution base fee, funds, 64 queued per sender, the pgas estimate) and a transaction the node already executed is refused without re-admission, so the pools converge and the chain's own blocks carry a transaction once. Dedup is by hash: the pool answers "known" for what it holds, executed or refused as invalid in the last 65,536 hashes, and one request per hash is outstanding across all peers. Per peer, 2,000 hashes a second with a burst of 8,192 are accepted in and served out; 4,096 hashes per message and 4 MiB per answer, over which the peer is dropped. A state-free fault (malformed, bad signature, wrong chain id, a refused type) disconnects the relaying peer, since every node refuses it the same way; a state-dependent refusal (nonce beyond the window, fee cap under the base fee, funds, queue depth, the 50,000-transaction pool cap) is dropped quietly, because the peer's tip may differ. Relay is off while the node is out of sync. Code: `protocol/flows/src/v10/evmrelay.rs`, the pump in `protocol/flows/src/service.rs`, the sink in `igneum/exec/src/service.rs` (`EvmTxRelaySink`).
Alternative. Per-block nonces or sequence-independent nonces (Sui-style objects). Rejected: every wallet assumes Ethereum nonces.
### 1.5 Invalid transactions are skipped by rule
@ -460,7 +462,7 @@ Rule change, 4 October 2026 (findings F-exec-A and F-exec-B of the attack suite,
| Units | 1 sompi = 1e10 wei; 1 IGN = 1e18 wei | Subsidies come from `igneum::block_subsidy` in 8-decimal sompi; the EVM is 18-decimal. The open "8 or 18 decimals" decision is unchanged; this is the fixed scaling at the bridge named there |
| Rewards | 80% of every blue block's subsidy to its miner, 20% to the proving pool escrow `0x...0220`, both credited in the segment that merges the block; reds unpaid | Design 4.4, with the pool held in a keyless account until proof records exist |
| Simnet | Devnet block rate and depths (1 BPS, k 18, mergeset 180, merge depth 3,600) with proof of work skipped; chain id 4463 shared with the devnet | A CPU test network of the devnet DAG shape |
| Mempool hand-out | A transaction handed to a template is not offered again for 4 s unless a chain block skipped it; a transaction skipped twice is dropped | Kaspa removes a block's transactions on block-added; the cooldown is the stand-in until the executor listens to block-added |
| Mempool hold | A transaction stays in every template until a block carrying it is added to the DAG (any block, this node's or a peer's: the executor subscribes to consensus `BlockAdded`); then it is held for 30 s or until the executor removes it (executed) or offers it again (a chain block skipped it); a transaction skipped twice is dropped | Kaspa's own rule (`mining/src/manager.rs`, `handle_new_block_transactions`). Replaced the 4-second hand-out cooldown on 5 October 2026 (fork `tx-gossip`, `pool.rs IN_BLOCK_HOLD`, `service.rs listen_block_added`): the cooldown made a sender mineable 1 s in 5 and inclusion came in 50-s bursts (bench-log, 5 October 2026 afternoon); with the hold a transaction is offered to every template until a block has it |
| Reorgs | Post-segment states for the last 64 chain blocks; deeper reorgs replay from genesis | Observed depth on the test network: 1 to 3 with Poisson-paced miners. A fixed per-template hold had made the three stub miners mine in lockstep rounds, and with equal work per block the GHOSTDAG hash tie-break then kept two equal-work chains alive from genesis (flips 48 deep every few seconds); `igneum-miner --hold-ms` is exponential now |
| Block tags | `pending`, `safe` and `finalized` all resolve to the executed tip | The virtual's segment is not executed eagerly and no certified checkpoint exists on this branch; the RPC does not pretend otherwise |
@ -474,7 +476,7 @@ Rule change, 4 October 2026 (findings F-exec-A and F-exec-B of the attack suite,
6. The virtual's segment is not executed eagerly (design 1.2 "about one second after inclusion"); the executor runs about one chain block behind the sink. `pending` tags resolve to the executed tip.
7. `eth_subscribe`, `debug_traceTransaction`, `trace_block`, `eth_getProof`, `eth_getUncle*`, `IgneumInfo`: not implemented.
8. The chain follower polls `get_virtual_chain_from_block` every 100 ms instead of subscribing to virtual-chain-changed notifications.
9. Mempool: no p2p relay of EVM transactions between nodes (each node's pool is what its RPC received), no eviction by age, no fee-based replacement beyond the 10% rule.
9. Mempool: p2p relay of EVM transactions implemented 5 October 2026 (section 1.4 "Relay", protocol version 14; measured on a 3-node fast-time chain A - B - C in the bench-log of that day: every transaction sent to A was included by B's and C's blocks). Still missing: eviction by age and fee-based replacement beyond the 10% rule.
10. Differential rows 1 (ethereum/tests), 3 (independent linearizer), 4 (Python oracle), 5 and 6 are not built; the harness here is the balance and receipt comparison of the acceptance criteria.
### 10.4 Merge plan with the finality branch

View file

@ -39,11 +39,11 @@ Versions in the table: `igneum-pow` is the Rust crate at `igneum-pow/Cargo.toml`
| 12 | The difficulty rule recovers from a hashrate step within minutes, where Kaspa's sampled rule never settles. A step inside an epoch set the rule oscillating on the live devnet on 4 October 2026; rule v2 removes it in the simulator and on a test network and is built but not yet rolled out | Spec 2.3; litepaper Speed (implied); bench page | tested by the team | repo `e9328c6`, `abb5a5d` (attacks), `67bf226` (rule v2); fork `difficulty` branch (timestamp fix) and `devnet-v4` `a21ff239` (`difficulty_v2_activation_daa`, `REF_WINDOW_V2 = 600`); `sim/difficulty/sim.py --live` | The live record `sim/difficulty/records/live-2026-10-04.csv` (8,090 headers, `pull_live.py`) and the hash-rate record beside it; `sim/difficulty/sim.py` on the synthetic set and the DAG replay; `sim/difficulty/attacks/attacks.py`; `sim/difficulty/testnet_v2.py` (3 nodes, activation at DAA 900); `cargo test --release -p kaspa-consensus --lib difficulty` (15 pass); bench-log "difficulty controller", "difficulty rule under attack", "timestamp attack fixed", "difficulty rule v2" | Live devnet v4, 4 October 2026 (UTC): a second RTX 5090 joining 7 minutes into an epoch (about 152 to 280 MH/s) hardened the difficulty 70M to 144M in 90 s and then swung by about a third for 40 minutes around the true level of 139M while the epoch-long reference lane carried the join; that card leaving for 4 minutes eased 116M to 67M and back to 106M; the epoch boundary with both PCs restarting took 152M to 77M in 3 minutes, after which the rule held within 1.3% per minute with no flips. Cause: the reference lane covered the whole epoch, so a mid-epoch step polluted it for the hour and the 25% trigger flipped on the short lane's noise. The DAG replay reproduces the record (std of log difficulty 0.115 against 0.134, 4.3 peaks against 4). Rule v2 (reference window 600 DAA) on the replay: std 0.026, 0 flips, mean 142.6M against 139M true; on a 3-node test network the v2 nodes eased a leave with no peak and held a rejoin within 3% after 60 s, and a node without the activation height forked off at it as designed. Rule v2 rolled onto the 12-node cloud network on 4 October (all nodes crossed the height on one chain; a hash-rate step then settled in 160 to 270 s with no swing) and activates on the devnet at DAA 33,000 the same evening. Timestamp forging (ledger M23) fixed the same day: a 50% forger drifts the rate under 1.1% where the 3 October rule gave it a 9.9x difficulty. Simulator, settled seconds: x50 step 62 to 66 (Kaspa 1,542), /50 step 657 to 753 (Kaspa 12,296). Apple M5 Max under load 7 to 442; the DAG model is fitted on one scale; the pool hopper's 0.7-point excess over Kaspa's rule stays open | none yet |
| 13 | Every node executes the ordered transactions natively and reaches the same state root | Litepaper Proving ("Every node executes ... natively"), Building ("runs on Igneum unchanged") | tested by the team | repo `f5f8c80`, `8dae48b`; fork `devnet-v4` `dc749905`; revm 43.0.3 | `node tools/evm-smoke/smoke.mjs` against a 3-node `igneumd`; `igneum-exec-diff seq.json`; bench-log "execution layer devnet v3" and "devnet-v4 integration" | Simnet, 3 October 2026: 87 of 87 viem checks, state roots identical on 3 nodes at four heights, 57 executed and 19 skipped transactions agree with plain revm, 0 mismatches. Merged node on real proof of work, 4 October 2026: 84 of 85 checks (the miss needs parallel blocks the network did not produce in 36 s), 59 transfers in 10 chain blocks, state roots identical on 3 nodes, `igneum-exec-diff` 0 mismatches over 59 transactions; the live devnet v4 runs this execution layer. Apple M5 Max. The prover is a stub; state is rebuilt from genesis at start; no EVM transaction relay between nodes | none yet |
| 14 | Ethereum bytecode runs unchanged, with the documented differences of spec 7.1 | Homepage Build card; litepaper Building | tested by the team | as row 13; fixes `F-exec-A`, `F-exec-B` (spec 7.5) | `tools/evm-smoke/smoke.mjs`: deploy via viem, `increment`, `hashLoop`, `eth_estimateGas`, `eth_getLogs`; `tools/exec-attacks` scenarios 1 and 3; bench-log "execution layer attack fixes" | Deployment, calls, reverts, logs and gas estimates behave as viem expects; chain id 4463; the prototype pgas table gives 0.0095 to 0.028 pgas per gas, below the design's band before calibration, 3 October 2026. 4 October 2026: a transaction that would cross the block's proving budget is refused by the mempool and, if forced in, aborted and charged with its nonce advanced (25 of 25 checks; 30 of 30 malformed cases). Apple M5 Max. The `Prover` precompile, proof records and the shard planner are not in the node | none yet |
| 15 | Every block is proven, with the proof landing within about a minute at launch | Homepage stats ("~60 s to a proof"); litepaper Proving; roadmap phase 3 gate | implemented | repo `d7e1f89` (GPU proof), `e01a3cc`, `292e800`, `eedd136` (`proving/igneum-prove`: shard cutter, MPT witnesses, shard and aggregator guests); SP1 6.8.1; spec 7.2, 7.6 | `proving/windows-wsl2` (SETUP-PROVER, PROVE-BLOCK) on the RTX 5090; `igneum-prove-host --mode block` on `proving/fixtures/`; bench-log "proving v0 on the RTX 5090" and "proving: devnet v4 shards" | First GPU proof of an Igneum block, 4 October 2026, RTX 5090 (WSL2, SP1 cuda, mining paused): fixture `block-78-increment` (2 transactions), core proof 1.4 s (7.3 MB, verify 0.221 s), compressed proof 2.7 s (1.27 MB, verify 0.038 s), post-state and receipts roots identical to the node's; 15.7x and 20.6x faster than a loaded M5 Max CPU. The same day on that CPU (load 38 to 47): a three-shard block proved shard by shard and aggregated by recursion, 19 min (1,139 s) end to end, 245 to 337 s per compressed shard proof, every proof verified. What is not there: no proof is produced, carried or checked on the chain (the devnet prover is a stub that signs claims), the proving pool pays nobody (row 21), the block proven is far below one shard, and the 60-second figure remains a design target; the pass mark is the standard in `docs/benchmarks/proving-e2e.md`. Second RTX 5090 run, 4 October 2026 evening (job run-20261004-173115): a full shard at the provisional S_p (6.75 M pgas, 60.8 M cycles) executed in 1.63 s, core proof 8.3 s (18.1 MB), compressed proof 10.9 s (1.27 MB, verify 0.040 s); a two-shard block (13.5 M pgas) proved shard by shard (11.7 s and 10.0 s) and aggregated in 2.2 s, 24 s of GPU stages end to end, every proof verified, six tampered witnesses rejected. The two host defects (an abort after the upload, an idle wait that turned out to be an unbuffered 18 MB proof save through the WSL2 file bridge, 24 minutes) are fixed (ledger P20) 5 October 2026, live devnet with real transactions (bench-log "real transactions, the first non-empty shard proven and paid"): block 72704 shard 0, 29 transfers, 5,800 pgas, proven on PC 2 in 34 s, verified on the Mac in 0.297 s and paid 1.7623 IGN, 53 s after the chain block executed; of about 1,400 blocks in the 20-minute window 36 were proven (the one prover takes the newest shard assigned to it), so "every block" is not yet true; a second content shard (72803, all copies skipped) failed the native-execution veto on the exporter's block structure, fixed with fixtures the same day, the node side pending the 0.3.9 rollout | none yet |
| 16 | A 12 GB card proves one shard in about 20 s | Litepaper Proving ("The proving budget"); roadmap gate 2 | designed | spec 5.1 (Target), 7.6 (`S_p` provisional, 7,500,000 pgas = `B_p` / 4) | `PROVE-SHARD.bat` on the RTX 5090 (pending); the end-to-end standard in `docs/benchmarks/proving-e2e.md`; bench-log "proving: devnet v4 shards" | Measured on a 32 GB card, not yet on a 12 GB card. A shard at the provisional `S_p` is 60.8 M SP1 cycles on the prototype pgas table (9 cycles per pgas, 44 per EVM gas; the modexp entry about 100x its SP1 cost); on an RTX 5090 (4 October 2026 evening, job run-20261004-173115) it executed in 1.63 s and its compressed proof took 10.9 s, verified in 0.040 s, so the 32 GB card is inside the 20 s target with margin. Whether a 12 GB card proves it at all, and in what time, is the next measurement (an RTX 3060 and an RTX 5060 Ti 16 GB are on order). A per-shard time can be met by shrinking the shard, so the project does not use it as a pass mark | none yet |
| 17 | The chip resistance target: a chip gains under 2x over a GPU | Litepaper Mining, "What Igneum does not claim"; homepage "no chip can be built for it" | designed | spec 0.2 (Target); O-1.17 | Public benchmark with a leaderboard by card model and a standing bounty, January 2027 (O-1.17); the on-die-SRAM test on the RTX 5090 (R3.5) | A target, not a measurement. Review round 3 priced a recompute chip with the 256 MiB cache on die at about 2.4x, approximate, before the usual chip-versus-GPU integer gain; the design answer (cache larger than any die) is open (spec 1.16) | none yet |
| 18 | The chip resistance measurements: the program is random-access bound, not bandwidth bound, and sits beyond a card's on-chip cache | Litepaper Mining ("bound by memory bandwidth", to be corrected), vs RandomX "Measured so far" | tested by the team | repo `aba248d`, `f2a1a64`, `4b95c5e` | RTX 5090 dataset sweep 4 MiB to 1 GiB with `proto-cuda/host.cu`; bench-log "RTX 5090 first run" and "dataset sweep" | At 1 GiB: 228.1 Mhash/s, 23.7 G random loads/s, 94.9 GB/s useful against a 1,638 GB/s dataset fill; inside the 96 MiB L2 (4 and 64 MiB) 1,340 to 1,353 Mhash/s, about 5.8x faster; 104 against 128 loads per hash gives 228 against 185 Mhash/s, proportional. 3 October 2026, RTX 5090, Windows, CUDA 12.8, version 1 programs. Prototype dataset 1 GiB against 2 GB at genesis; a pure random-read microbenchmark (R3 chip designer, attack 2) has not run; the sweep has not been repeated on version 2 | none yet |
| 19 | The lottery hash is sound as a hash: uniform output, deterministic, no out-of-bounds read, fuzzed | Litepaper vs RandomX ("Every number above is measured and logged") | tested by the team | repo `c52307e`, `58a5a63`, `b27da39`; `proto-metal/TESTS.md` | `proto-metal/igneum-bench --fuzz --edge --stats --determinism --memcheck`; `--fuzz 2000` on the version 2 generator; `igneum-census`; bench-log "hardening tests", the re-run on the memory-hard dataset, "generator version 2 adopted" | Version 1: 10,200 random programs, 1,305,600 hashes, 0 mismatches; 14 of 14 edge cases; bit frequency within 2.90 sigma, avalanche mean 31.99 to 32.04 of 32; deterministic fingerprint across 5 runs; every dataset read masked, 3 October 2026. Version 2, 4 October 2026: 2,000 random programs through the Metal cross-check, 8,000 warps, 0 mismatches, 128 loads per hash on every program; 20,000-program census, 5.2% rejected (4.1% static, 1.1% dynamic). Apple M5 Max. Statistics are not a security proof; the edge, stats and memcheck sections were not re-run on version 2 (they do not depend on the generator); the seed derivation review (O-1.4) is open; the fuzz set has run on Metal and the CPU only | none yet |
| 15 | Every block is proven, with the proof landing within about a minute at launch | Homepage stats ("~60 s to a proof"); litepaper Proving; roadmap phase 3 gate | implemented | repo `d7e1f89` (GPU proof), `e01a3cc`, `292e800`, `eedd136` (`proving/igneum-prove`: shard cutter, MPT witnesses, shard and aggregator guests); SP1 6.8.1; spec 7.2, 7.6 | `proving/windows-wsl2` (SETUP-PROVER, PROVE-BLOCK) on the RTX 5090; `igneum-prove-host --mode block` on `proving/fixtures/`; bench-log "proving v0 on the RTX 5090" and "proving: devnet v4 shards" | First GPU proof of an Igneum block, 4 October 2026, RTX 5090 (WSL2, SP1 cuda, mining paused): fixture `block-78-increment` (2 transactions), core proof 1.4 s (7.3 MB, verify 0.221 s), compressed proof 2.7 s (1.27 MB, verify 0.038 s), post-state and receipts roots identical to the node's; 15.7x and 20.6x faster than a loaded M5 Max CPU. The same day on that CPU (load 38 to 47): a three-shard block proved shard by shard and aggregated by recursion, 19 min (1,139 s) end to end, 245 to 337 s per compressed shard proof, every proof verified. What is not there: no proof is produced, carried or checked on the chain (the devnet prover is a stub that signs claims), the proving pool pays nobody (row 21), the block proven is far below one shard, and the 60-second figure remains a design target; the pass mark is the standard in `docs/benchmarks/proving-e2e.md`. Second RTX 5090 run, 4 October 2026 evening (job run-20261004-173115): a full shard at the provisional S_p (6.75 M pgas, 60.8 M cycles) executed in 1.63 s, core proof 8.3 s (18.1 MB), compressed proof 10.9 s (1.27 MB, verify 0.040 s); a two-shard block (13.5 M pgas) proved shard by shard (11.7 s and 10.0 s) and aggregated in 2.2 s, 24 s of GPU stages end to end, every proof verified, six tampered witnesses rejected. The two host defects (an abort after the upload, an idle wait that turned out to be an unbuffered 18 MB proof save through the WSL2 file bridge, 24 minutes) are fixed (ledger P20) 5 October 2026, live devnet with real transactions (bench-log "real transactions, the first non-empty shard proven and paid"): block 72704 shard 0, 29 transfers, 5,800 pgas, proven on PC 2 in 34 s, verified on the Mac in 0.297 s and paid 1.7623 IGN, 53 s after the chain block executed; of about 1,400 blocks in the 20-minute window 36 were proven (the one prover takes the newest shard assigned to it), so "every block" is not yet true; a second content shard (72803, all copies skipped) failed the native-execution veto on the exporter's block structure, fixed with fixtures the same day, the node side pending the 0.3.9 rollout 5 October 2026, evening (bench-log "proving v1"): the aggregated segment record, the chain rule and the unproven rule are implemented behind `proving_v1_activation_daa` (branch proving-v1, not on the devnet before 0.3.11); on the RTX 5090 a chain of 8 consecutive live blocks proved and aggregated by recursion in 135.6 s with the miner on the card (17 s a block, one proof of 1,272,909 bytes attesting all 8, verified in 0.04 s); the 3-node fast-time harness paid a segment record 1.0 s after submission and refused a late one after its deadline (21 checks); the devnet itself, with one prover, carried proofs for 2.4% of blocks over 30 minutes at a block-to-record latency p50 44 s, p99 52 s. The "within about a minute" holds per proven block; "every block" needs 18 mining 5090s or 6 proving-only cards at empty blocks on the measured rates, and the mandatory rule stays off until the share is one | none yet |
| 16 | A 12 GB card proves one shard in about 20 s (WITHDRAWN 5 October 2026: a 24 GB card proves a full shard at the adopted size in 4.3 s; 32 GB mines and proves) | Litepaper Proving ("The proving budget"); roadmap gate 2 | designed | spec 5.1 (Target), 7.6 (`S_p` provisional, 7,500,000 pgas = `B_p` / 4) | `PROVE-SHARD.bat` on the RTX 5090 (pending); the end-to-end standard in `docs/benchmarks/proving-e2e.md`; bench-log "proving: devnet v4 shards" | Measured on a 32 GB card, not yet on a 12 GB card. A shard at the provisional `S_p` is 60.8 M SP1 cycles on the prototype pgas table (9 cycles per pgas, 44 per EVM gas; the modexp entry about 100x its SP1 cost); on an RTX 5090 (4 October 2026 evening, job run-20261004-173115) it executed in 1.63 s and its compressed proof took 10.9 s, verified in 0.040 s, so the 32 GB card is inside the 20 s target with margin. Whether a 12 GB card proves it at all, and in what time, is the next measurement (an RTX 3060 and an RTX 5060 Ti 16 GB are on order). A per-shard time can be met by shrinking the shard, so the project does not use it as a pass mark 5 October 2026, evening (bench-log "proving v1", the S_p curve): measured on the RTX 5090 with SP1 6.8.1's GPU prover, the card to itself, 1-s nvidia-smi samples: an empty shard 13,874 MiB and 2.2 s; a full shard at the ADOPTED v1 budget (30,000 pgas, 4.7 M cycles) 20,434 MiB and 4.3 s; the full prototype shard (6.75 M pgas, 60 M cycles) 28,307 MiB and 10.8 s; beside the miner 15,670 and 30,039 MiB. No environment knob of SP1 moves the 13.9 GB floor and the GPU server has no options of its own, so on this build a 12 GB card proves nothing, a 16 GB card only empty shards, a 24 GB card the adopted full shard alone and beside the miner (22,210 MiB and 13.2 s, measured on the 32 GB card: the 5090's allocation pattern, not yet a run on a 24 GB card) and a 32 GB card the prototype shard beside the miner with 2.5 GB spare. The litepaper line now says so; the 12 GB gate returns when a prover build with a smaller floor is measured on a 12 GB card | none yet |
| 17 | The chip resistance target: a chip gains under 2x over a GPU | Homepage hero and litepaper abstract ("a custom chip gains under 2x, and the model and the bounty are public"), litepaper "What Igneum does not claim" | tested by the team (the model), designed (the target) | program class v3 (Counter ASIC 2.0, 5 October 2026): branches ca2-v3 d233fa1 and after, ca2-mixer 1ab8b21, ca2-era 78c0ee4; `docs/analysis/chip-model-v3.md`, `docs/analysis/sram-mirror.md`, `docs/analysis/scratch-soundness.md` | The m16 recompute model re-run on the measured v3 rates and verifier times; the on-die-cache chip row | The on-die-cache recompute chip against the RTX 5090's measured 136.1 MH/s: class v2 2.4x; class v3 (mixer x8) 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon; margin 8% on the allowance, 9% on the budget. 5 October 2026, M5 Max, RTX 5090, RX 9070 XT. The 2x target is a target: no chip has been built; the bounty stands (O-1.17) | none yet |
| 18 | The chip resistance measurements: the program is latency-bound (random reads), not bandwidth-bound, on every card we own, and sits beyond a card's on-chip cache | Litepaper Mining ("waits on memory latency, not on maths or bandwidth"), vs RandomX; the numbers page | tested by the team | readwidth e752fc7 (`docs/plans/read-width.md`), ca2-era 78c0ee4, ca2-cache 2de19e5 (`docs/plans/hot-table.md`) | The dependent-read probes at 32 to 1,024 MiB and the hash rate per class on the three cards; the latency-bound share = rate over the probe ceiling per load | Latency-bound share at the 1 GiB dataset: RTX 5090 0.96 (v2) and 1.01 (v3), RX 9070 XT 0.87 and 0.95, M5 Max 1.01 and 1.06; wider reads do not close the AMD gap (the 9070 XT does 2.4 G dependent reads per second at every width; the 5090 goes bandwidth-bound at 64 B, share 0.58); a 32 to 96 MiB hot table is not kept resident by any card while the dataset streams (g 0.80 to 0.87 in the added form). 5 October 2026 | none yet |
| 19 | The lottery hash is sound as a hash: uniform output, deterministic, no out-of-bounds read, fuzzed; class v3 bit-exact on the three vendors | Litepaper vs RandomX ("Every number above is measured and logged"), the numbers page | tested by the team | ca2-mixer 1ab8b21 (`tests/mixer.rs`, `tests/scratch.rs`), ca2-era 78c0ee4, ca2-soundness a465881 (`docs/analysis/scratch-soundness.md`), `igneum-pow/tests/packs.rs` | The crate suite (53 + 4 + 19 + 7), the Metal fuzz, edge, stats and determinism runs on the v3 construction, the pack vectors and 2^24 fingerprints on Metal, Apple OpenCL, the RTX 5090 and the RX 9070 XT, the 1,024-hash CPU re-check per card | Class v3 (mixer x8 + era): 200-program fuzz 200 of 200 on Metal, every tenth on Apple OpenCL; the pinned v3 packs 3/3 + 3/3 and 96 of 96 lanes on Metal and Apple OpenCL; the six era packs' fingerprints equal on the three vendors (PC 1 job run-ca2-era-pc1-20261005, 5 October 2026); the v2 exports byte-identical on the v3 crate; the final-class PC rows and the G2 re-check: job run-ca2-era-pc1b-20261005 (pending at the time of writing) | none yet |
| 20 | No premine, no pre-sale, no allocation: every coin is minted by the schedule and every coin goes to the block producer (80%) and the proving pool (20%) | Homepage stats and Economics tiles; litepaper Supply, Economics | implemented | repo `6ac80a3`; fork "igneum-node devnet v0"; `consensus/core/src/igneum.rs`, `coinbase.rs` | `cargo test -p kaspa-consensus-core igneum` (8 pass: subsidy table, ramp, split, cap) and `cargo test -p kaspa-consensus coinbase` (8 pass); `igneum-miner inspect 40`; bench-log "igneum-node devnet v0" | Coinbases on the devnet: 80/20 exact on 39 of 39 single-payee blocks, the 20% to the `igneum-proving-pool-v0` output; the per-second schedule sums to under the 4,000,000,000 cap by less than 100 coins; 3,168,808,781 units per DAA second in years 0 to 2, halving at 63,115,200 DAA s. 3 October 2026, Apple M5 Max. The devnet genesis carries no allocation; the mainnet genesis does not exist yet, so the claim is about the code and the stated rule, not a launch that has happened | none yet |
| 21 | The proving pool's 20% reaches shard provers and aggregators | Litepaper Economics; homepage "20% provers" | tested by the team | spec 5.3; `proving/igneum-prove` carries the prover's payout address in every shard proof (ledger P12) | None. The pool output exists (row 20); the payout from it against proof records is unwritten. Since 5 October 2026: the payout rule is live on the devnet (`proving.rs shard_payouts`, the carrying segment pays the first valid record per shard its part of the segment's pool credit) | The escrow accumulated on the simnet (92.55 IGN at the end of the v3 run) and nothing can draw it. Rule decided: per block, divided among shards by consensus proving cost, sortition to 8 provers for 10 s then open (spec 7.2). The economy model of 4 October 2026 (`sim/economy`, 1,000 operators, 30 days) kept every block proven within 60 s under six stress scenarios; a model, not hardware Live devnet, 5 October 2026: 388 shards paid by 16:02 UTC, 446.13 IGN from the pool to PC 2's payout address, 0.8813 IGN per mergeset block of the proven segment (bench-log entries of 5 October: "the first shards proven, verified and paid" and "real transactions, the first non-empty shard proven and paid") | none yet |
| 22 | The base fee is burned in full and the priority fee splits 80% to the miner and provers, 20% to the apps whose code ran | Homepage Economics caption and Build card; litepaper "Where fees go" | tested by the team | repo `f5f8c80`; fork worktree `vendor/igneum-node-exec` | `tools/evm-smoke/smoke.mjs` receipt checks; bench-log "execution layer devnet v3" | Transfer receipt: `burnedProvingFee` 200 gwei, `minerTip` 16,800 gwei (80%), unregistered developer share 4,200 gwei burned; contract call: 80% to the miner, 20% credited to the payee the constructor registered, balance delta equal. 3 October 2026, Apple M5 Max simnet. The provers' part of the 80% is not split out (no provers exist); the base fee stayed at the 1 gwei floor throughout | none yet |
@ -87,6 +87,7 @@ Versions in the table: `igneum-pow` is the Rust crate at `igneum-pow/Cargo.toml`
|---|---|---|---|
| 15 | implemented | implemented, with a live result | the first non-empty shard (block 72704, 29 transfers) proven, verified and paid on the devnet; not every block is proven yet |
| 21 | designed | tested by the team | 388 shards paid from the pool on the live devnet, the rule in `proving.rs`, the numbers in the bench log |
| 22 | Card lifetime: a 4 GB card mines about four years and an 8 GB card about twelve, under the dataset's step schedule (2 GB at genesis, doubling at years 4, 12, 28, 60) with the cache freed after the daily build | Litepaper Hardware and vs RandomX ("Dataset" row); homepage Mine card and "Memory" row | designed | `docs/analysis/card-lifetime-2026-10-05.md` (branch card-lifetime 1fecfe2); spec 1.13.3 option (b) recommended to the project lead 5 October 2026 (`docs/plans/counter-asic-2-rollout.md` 6c) | The per-tier working-set arithmetic of that document (GTX 1650, RTX 3050, RTX 3060, RTX 4090 tiers) against the step schedule | A design claim: under the continuous mapping (a) a 4 GB card is out within 1 to 1.5 years and an 8 GB card at 6 to 7.5 years, so the sentence is true only under the step schedule (b), which the spec has not yet fixed (O-1.13) | none yet |
## What would move a row

View file

@ -621,6 +621,8 @@ Evidence: design doc Finality v2, Fork choice items 1 to 4; `sim/results.md` fin
Sweep (5 October 2026, evening): the module-on against module-off comparison of O-3.8, run on the fast-time harness with the live node line (`tools/finality-attacks/c4.mjs`, fork 2b6d23ef, 3 nodes, 100-ms proxied links; raw tables in `docs/bench-log.md`, "FUD ledger sweep round 6", C4). The scenario separates weight from work: side B (n1, n2, four keys) holds 70% of the weight table and side A (n0, two keys) 30% when the link is cut; from the cut A mines at 0.6 blocks/s and B at 0.4, so A's chain is the heavier one by blue work while only B can certify under rule v3 (A holds 30% of the frozen table). Module off (`min_daa` never, so no certificate can form, fork choice bare GHOSTDAG): after a 150-s split the three nodes converged on A's heavier chain within 36 s of the heal, B's nodes re-determined their two split-time checkpoints onto it (F24), 0 conflicts. Module on (rule v3 from checkpoint DAA 0), 90-s split, n0 back on the link 6 s after the heal, A's chain at about 58 DAA of its own time, well inside the 120-DAA frozen table: during the split A locked nothing and B locked indices 7 and 8 on its own blocks, as designed; after the heal n0 did not switch. Its log: B's certificates for 8 and 9 arrived and were "kept pending until the chain decides (no lock at this index)" (the F24 path), n0's chain never changed because GHOSTDAG prefers its heavier tip and nothing in the node turns a verified certificate over an off-chain block into a fork-choice constraint, and one window after n0's last lock (index 7 at DAA 209, so from DAA 329) the frozen table no longer applied on A's chain ("no frozen table (no lock on this chain inside the window)"), A's two keys were 100% of A's own window table (B's post-cut blocks are red there and earn nothing), and n0 locked 10, 11 and 12 alone; B's certificates for 10 and 11 then logged CONFLICTING on n0, and B's nodes kept their certified chain. End state: sinks apart, 2 locked indices disagreeing across the nodes, a finality fork from a 96-s honest partition with no attacker and the frozen table intact at the heal; the same shape with a 150-s split (the table expired at the heal) and under rule v2 (the control: 1 conflict, sinks apart). So the answer to the critic is sharper than conceded: the overlay is specified to override blue work (spec 3.5, "GHOSTDAG among tips through all certified checkpoints") but the shipped node applies a certificate only to a block on its own chain, holds the rest pending a reorg that GHOSTDAG alone never produces, and after one window the heavier side certifies its own chain. Two honest views never reconcile. What closes it: a verified certificate over a block the node does not have on its selected chain must verify against the weight table at THAT block (its signers' weight there) and, when valid, constrain fork choice to tips through it, forcing the reorg (a certificate-driven reorg, bounded by the finality depth), with the node's own unlocked records re-determined on the new chain (F24); until then the exchange guidance of 3.9 (a node partitioned for more than a minute treats its locks as proof of work until it has seen the network's certificates agree with its own) is the only protection, and the 3.11.7 row for this case ("a certificate over a chain the node is not on") is missing. On the live devnet the window is 7,200 DAA (two hours) and the cliff is two hours after a side's last lock; a miner who joins with more hashrate than the weight table credits is the realistic work-majority side. The trace-driven adversary of O-3.8 is still owed. Spec rows: 3.5, 3.11.4, 3.11.7; node: `processes/finality.rs` (`ingest_certificate`'s pending branch, `fork_choice_lock`). Decision owner: the project lead (gate 3; a rule change to the node's fork choice).
Fix (5 October 2026, night): the certificate-driven reorg, built, unit-tested and measured; fork branch `c4-fix` on release-0.3.6 (a24ab01a), main branch `c4-fix`. The cause in the code: `processes/finality.rs` `ingest_certificate` verified a certificate only over the node's own determination and sent every other block to `hold_pending`; `fork_choice_lock` reads `state.locks`, which only `evaluate` filled over the node's own chain; so a certified block off the chain never became a lock. A second cause only the harness showed: the block relay (`protocol/flows/src/v10/blockrelay/flow.rs`) skips a relayed block lighter than the virtual's merge-depth root, and the certified chain is the lighter one by construction, so the heavier side never even received it. The fix: `ingest_off_chain` verifies a certificate against the voter table at its own block (canonical list, aggregate BLS, 2/3 of active and of total there, the frozen table under v3, the first-month gate), checks the block lies on the chain through the node's nearest locks (else CONFLICTING, 3.11.4, no lock withdrawn), locks the index on that block and asks the virtual processor to resolve (`VirtualStateProcessingMessage::Resolve`), so the sink search keeps only tips through it, whatever the blue work and whatever the merge depth (finality outranks merge depth; the depth-based finality point still bounds it, Kaspa's pruning safety, logged once); pending certificates over blocks the node lacks are retried on every virtual change; `evaluate` locks the node's own determination only on the chain through its locks; and while a pending certificate names a block the node lacks (`finality_wants_blocks`), the relay takes the lighter block, which orphans, falls out of range and triggers IBD of the certified chain. Not gated on v3: the live devnet's rule v2 took the same pending path (unit test `the_certificate_driven_reorg_holds_under_rule_v2`). Spec 3.5 carries the rule in one paragraph, 3.2 C4 and the 3.10 rows C4 and F1/F2 the implementation. Measured (bench-log "the C4 fix", `c4.mjs` with `WINDOW=240` so a 130-s split plus the 84-s p2p reconnect stays inside the window; the sweep's 120-DAA framing crosses F21's bound before any certificate can arrive once the real reconnect time is counted): under rule v2 side B locked index 12 during the split, n0 took B's first relayed block through the hook, locked 12, 13 and 14 by certificate within 2 s, re-determined 11, and all three nodes ended on B's certified chain with 0 CONFLICTING and 0 disagreeing locked indices (was: sinks apart, 5 conflicts on each node, 2 disagreeing); the module-off control is unchanged (heavier chain, 0 conflicts); under rule v3 (frozen table on, 140-s split, n0 reconnected 3 s after the heal) B locked 11 and 12 during the split, n0 adopted 12 by certificate and verified 11 on the new chain, all three nodes on B's chain, 0 CONFLICTING, 0 disagreeing; the mirror case (B certified nothing, A certified after the heal) had B's nodes adopt A's certificates and move before IBD. Unit tests on PC 2: `kaspa-consensus` 97 passed, `kaspa-consensus-core` 101 passed. Still owed: the live-devnet partition test of O-3.6 and the trace-driven adversary of O-3.8. What the fix does not cover, by design: a partition that outlasts the bound before the certificate arrives (the side has locked alone, 3.11.4 keeps it, the late certificate is CONFLICTING for the operator), which on the devnet means over two hours. Rollout: a consensus-behaviour change in the node with no params-digest change; a mixed fleet disagrees only in the state the old node already got wrong (an old node holds the certificate pending and stays on its heavier chain while new nodes move), and converges once every node is new; ship in the next node release with every node restarted on it.
### C5. vs Ethereum: you compare inclusion to finality
"'Included in about one second, against twelve on Ethereum.' Inclusion in a DAG is not confirmation. Ethereum's twelve seconds is a slot, its finality is about thirteen minutes, and you compare your two-minute lock to that as if a two-minute lock by a pool committee were the same thing."
@ -1255,6 +1257,8 @@ Sweep (5 October 2026, evening): the two options, with their measured cost.
Recommendation: Option B, which spec 3.11.4 already states and O-3.17 names; it is Kaspa's rule for a finality conflict (`vendor/rusty-kaspa/consensus/notify/src/notification.rs`, `FinalityConflict`), it is the only reading under which an exchange can credit on a lock, and its cost falls on a state that needs a 34% equivocator or a 30-day partition. What it needs: the 3.5 paragraph replaced by 3.11.4's text, `finality_conflict` and the `finality_active` clear in the node, and the forced-double-certificate devnet test of 3.11.7. Decision owner: the project lead (gate 3).
What the C4 fix changes for option B (5 October 2026, night): the honest-partition row above is gone. Before the fix a 96-s partition with the table intact put two certified chains on the network with no equivocator (C4: the work-majority side held the other side's certificates pending, then certified its own chain), and option B would have paused finality on every node of that side for an operator. With the certificate-driven reorg (spec 3.5, `ingest_off_chain`) a node that receives a valid certificate for a chain it is not on adopts it and moves, so after a heal shorter than a window there is one chain of locks and nothing to withdraw: the module-on harness ended with 0 conflicting certificates and 0 disagreeing locked indices on all three nodes, under rule v3 and under v2 (bench-log "the C4 fix"). What remains for option B is exactly the states 3.11.4 names: an equivocator at one third or more, and a partition longer than a window (both sides certify their own chain before the heal; the node then holds a lock at a higher index on the other chain, `off_lock_chain`, and the late certificate is CONFLICTING, kept for the operator, no lock withdrawn). The node still does not clear `finality_active` or expose `finality_conflict` (O-3.17).
Answer: Correct as the proposal stands. For an exchange "locked" must be irrevocable or it is a confirmation count. The alternative is Kaspa's: a verified certificate is never re-evaluated; two certificates at one index are a chain split that halts `finality_active` until an operator intervenes, and the node never reports a lock it may withdraw. Equivocation costing history and not coins (F6) means the attacker who caused the split keeps the deposit either way. Decision at gate 3; the devnet partition-and-heal test of O-3.6 measures whichever rule is chosen.
Evidence: spec 3.5, 3.9. Review id R3.17.

View file

@ -0,0 +1,75 @@
# Consequences ledger, 5 to 6 October 2026 (night)
The standing consequences reviewer (CLAUDE.md, "Every number carries its consequences"). One row per number whose consequence for a user tier, a chip builder or a public claim nobody had stated or acted on. Tiers: a home miner with one 8, 12, 16, 24 or 32 GB card; a rig; a pool user; Windows, Linux, macOS; NVIDIA, AMD, Apple. Times UTC. State: open, sent (the owner has the message), in work, closed, decision (in `consequences-decisions.md`).
Rows already handled before this ledger opened, for the shape: 15.6 GB mine-and-prove peak (12 and 16 GB profiles, the sweep `memsweep-pc2-pv1`); 9070 XT 18 MH/s (the read-width experiment); the cache never grows (layer 6 option C); aggregation 9.7 s a block (the aggregation-cost agent).
## Round 1 (21:30 to 22:30)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C1 | Fee switch H = 210,000, reached about 18:45Z on 6 October (DAA 137,041 at 22:32:54Z, 1.002 DAA/s averaged since the 15:40Z read; the plans first said 19:50 from 0.965 blocks/s) | `docs/plans/fee-switch-devnet.md` sections 3 and 7; `vendor/igneum-node-0310/igneum/exec/src/rpc.rs` 849 (no `daaScore`, no `feesV1ActivationDaa` in `igneum_exportSegments`); the app's exporter call without `--fees-v1-activation-daa` (`app/igneum-app/src/prover.rs` 516 to 519) | every prover on the devnet (PC 2, the Mac, any 0.3.10 machine) | From the first chain block at or above H the 0.3.10 node cuts 30,000-pgas shards and meters with the v1 table, but the app's exporter sees a dump without the switch and cuts at 7.5 M with the prototype table: the host refuses the fixture (wrong `S_p`) or the node vetoes the statement. Proving on the devnet goes dark at H and nothing is paid until every prover runs a node whose export carries the switch (the proving-v1 fork does, `vendor/igneum-node-pv1` rpc.rs 961 and 982). `pc2-chain.ps1` fixtures from the live node fail the same way after H | Either 0.3.11 (proving-v1 fork) on every prover before H, or the switch republished at a later H before 19:50Z tomorrow (a digest flip, every node). Recommendation in `consequences-decisions.md` D1 | proving v1 acd4f36bc2c07a4e2, shipper ae892a8b0f78fe31c, coordinator ada8afb62d752b1e2 | in work (proving agent: fork eb32c645 on 21d4c73c carries daaScore and feesV1ActivationDaa, the exporter needs no flag; 0.3.11 on every prover before 16:00Z on 6 October or H moves to tip + 86,400; the line is in the proving plan, the rollout plan and release-0.3.10.md) |
| C2 | 16,751 MiB GPU peak during the chain of 8 with the miner resident (`chain-pc2-pv1c`) | bench-log "proving v1", step 2 chain row | 16 GB cards (RTX 5080, 5060 Ti 16 GB, 4060 Ti 16 GB) | The plan's "a 16 GB card sits 0.4 GB under tonight's peak" is the empty-shard row (15,590 MiB). The chained aggregation adds 1.2 GB and lands at 16.4 GB, over a 16 GB card. So a 16 GB card cannot mine and aggregate on this build; it can at most mine and prove shards, with 0.4 GB spare and no full shard measured | The 16 GB chained-aggregation row goes into the sweep; the aggregator step in `prover.rs aggregate_once` gates on card memory (24 GB with the miner running, else pause the miner on that card for the aggregation); the Proving tile says which role the card runs | proving v1 | closed as measured (proving agent, spcurve-miner-pc2-pv1 219517f: the adopted shard beside the miner 22,210 MiB and 13.2 s, so a 24 GB card has 2.3 GB spare from the fee switch, the number approximate for the card itself because it is the 5090's allocation pattern; the prototype shard 30.1 GB, the 32 GB card alone; aggregation_card gates on the same memory rule, 23,552 MB either role). Open: the first real 24 GB card measurement, and the public line says "24 GB" from a 32 GB card's pattern (D2 wording) |
| C3 | 13,816 MiB GPU peak, prover alone, empty shards (`prover-cost-pc2-pv1`) | bench-log "proving v1", step 1 | 12 GB cards (RTX 3060 12 GB, 4070, 5070), the default-on rule (`provedefault.rs` `MIN_VRAM_MB` 11,776) | On the only measurement the prover by itself exceeds a 12 GB card by 1.5 GB, so the 12 GB default gate switches proving on for cards that cannot run it on this build unless the SP1 knobs bring the peak down. The litepaper's "12 GB or more proves full shards" (`site/litepaper.html` 560) and evidence row 16 rest on the sweep | 0.3.11 does not ship the default-on until the sweep has a row under 11.5 GB for a full v1 shard; if none, the gate moves to 24 GB and the public claim is qualified (D2) | proving v1 (sweep in work); public claim: decision | closed as measured (proving agent: the sweep moves no floor, 13.9 GB for an empty shard, 28.3 GB for a full prototype shard; the default is 20 GB mining / 16 GB prove-only; the 12 GB sentence is false on this build and goes to the project lead as D2 with the curve); the v1-shard row is C15 |
| C4 | Host RAM 25,550 MB used of 63,132 on PC 2 with the prover on; the WSL2 VM working set 7,915 MB | bench-log "proving v1", step 1 host RAM row | Windows home miners with 16 GB RAM (the common gaming PC); Macs with 16 GB switching the CPU prover on | The prover default has no RAM floor. On Windows the WSL2 VM alone holds 7.9 GB beside the app, the node and the game-class desktop; a 16 GB machine with proving on by default swaps or kills the node. The app's own RSS (engine, node, verifier) is not separated in the measurement, so no requirement can be stated yet | The sweep job records the app's and the node's working sets beside the VM's; `provedefault` reads total RAM and stays off under 32 GB on Windows until measured; the Proving tile and the miner page state the RAM requirement | proving v1; miner UI adbf58b058186a18b (the tile line) | closed in code (proving v1 c2544be: MIN_RAM_MB_WINDOWS 31,000 in provedefault.rs, the tile line names the 7.9 GB VM on a 63 GB PC; unknown RAM is not a gate; the miner UI help line next cut) |
| C5 | The app quit for the 0.3.10 update at 20:01:09Z and aborted the chain job at block 3 ("aborted (the app is quitting)"); `prover.rs` kills the child on quit | bench-log "proving v1" chain run 1; `app/igneum-app/src/prover.rs` 266 to 272 | every prover on every update; with proving v1 the aggregator | Tonight's `update-now` to every app aborts whichever shard each prover has in flight (up to 37 s of work each, no payout, re-assigned to nobody until the window passes). Under proving v1 a restart mid-segment loses the aggregator's chain state: the segment goes unproven after T (600 DAA), the aggregator share of 8 blocks is forfeited to the escrow, and the next record must be fresh-chain. With one aggregator on the devnet every app update costs 10 minutes of unproven segments | The update's safe moment waits for the prover's current proof (as it does for the node); the aggregator persists the last segment proof and resumes the chain after a restart; the Updates section says "waits for the proof in flight". For 0.3.10 tonight the aborted shards are an accepted cost, recorded | proving v1; shipper (tonight's rollout note); miner UI (Updates wording) | refused for tonight by the proving agent (persisting the segment proof and holding the update for a proof in flight are 0.3.12; the cost is in the plan as the rule working as written); the shipper carries the aborted-shard count in release-0.3.10.md section 8; the 20:01:09Z abort was NOT 0.3.10 (shipper: nothing published), see C16. The count, read from the intake at 22:3xZ: PC 2 quit for the 0.3.10 restart at 21:49:29Z (run win-1ccfe586-20261005-200114) with no proof in flight, because its prover had been dark on the root socket since 21:25Z (last exporter line 21:44:02Z, C17); the Mac proves nothing by default; so the restart aborted 0 shards tonight and the first instance of this class will be the 0.3.11 rollout |
| C6 | The prover costs a mining 5090 4.0% (124.7 to 119.7 MH/s); the software dev fee is 1 template in 100 | bench-log "proving v1" step 1; `packaging/hive/README.md` "The dev fee" | pool users; the pool's ledger | A member who proves sends 4% fewer shares, so the pool's vardiff and `stats.hashrate` read a 4% loss while the proving income (90% of the shard credit plus a tenth to an aggregator) is paid to the member's own key and never appears in the pool's ledger: the pool dashboard understates a proving member's earnings. The software dev fee has no mechanism in pool mode (the pool issues the templates), so a pooled miner pays no dev fee today and the pool design must say whether it takes one (1 share in 100 to the dev address) or none | The pool-v0 design states both: the member `stats` carry a `proving` flag and the pool page shows proving income beside shares; the dev fee rule in pool mode is written down before the first pool ships | pool a4781ba117326091f | closed on pool-v0 (pool agent: the member stats line carries a proving flag, /api/miners/<address> and the pool page show it with the 4% note and that proving income never passes through the pool; the software dev fee is NONE in pool mode, the pool's own fee (default 1%) is the only fee, carried in the welcome message's share_scheme, shown on the Connect card, written in docs/plans/pool.md and packaging/hive/README.md). Merge note: packaging/hive/README.md is now edited on three branches (hive-words, ota-k2, pool-v0). Public wording: the miner page's "a visible 1% software fee you can switch off" is a solo-mining sentence once a pool exists, added to D8 |
| C7 | The HiveOS package holds `igneumd`, `igneum-miner` and the two workers; no `igneum-prove-host`, no SP1 GPU server, no per-card rule | `packaging/hive/make-hive-package.sh` 26 to 30; README "What the hooks do" | rigs (HiveOS, Linux), the largest hashrate tier | A rig cannot prove at all, so the 20% proving share is reachable only from the app. The miner page says "the card mines and proves" (`site/miner.html` 7, `site/index.html` 484) and the HiveOS README does not say rigs mine only. A rig that could prove needs the 251 MB SP1 GPU server, CUDA 12.8 and the per-card profile of C2 and C3, and a rig's RAM (4 to 8 GB on most Hive images, approximate) is below C4's floor | The rig installer either carries the prover with the per-card rule and a RAM check, or its README and the miners page say rigs mine only and provers are app machines; `IDENTITIES=8 for a big card, 2 for a small one` gets a threshold in GB | rig installer a3e7b2b03222f5cff | closed (rig installer 88f31d9: gates 23,552 MB for both roles from provedefault.rs 440fd59, the README table per tier; HiveOS hive-words 2d056e8: the same table; open only the miners page sentence, D8) |
| C8 | Dataset 2 GiB at genesis plus 0.5 GiB a year; the working-set rule "under 6 GB on an 8 GB card"; the cache 256 MiB doubling at years 4 and 12; scratch up to 128 KiB per resident warp | spec 01 section 1.13.3; coordinator's budget rule; layer 6 option C; `docs/plans/hot-table.md` section 3 | 4 GB and 8 GB cards; the litepaper's claim | `site/index.html` 461 says "Any 4 GB card" and the litepaper (560) says a 4 GB card mines for about four years and an 8 GB card for more than a decade. At 75% of the card the 4 GB card's dataset room is about 2.4 GiB: under one year. The 8 GB card's room is about 5 GiB after cache, hot table, scratch and buffers: about six years, five and a half with the year-4 cache step. Neither public sentence holds under the schedule | A card-lifetime table per tier (sub-agent, `docs/analysis/card-lifetime-2026-10-05.md`); the public wording is a claim for the project lead (D3) | sub-agent (table); decision (wording) | table landed (sub-agent, docs/analysis/card-lifetime-2026-10-05.md, 1fecfe2, merged into ca2-coord at 22:50): 4 GB 1.0 to 1.5 years under (a) and year 4 under (b); 8 GB 6.3 to 7.5 or 12; 12 GB 12 to 13.5 or 12 to 28 depending on whether the cache stays resident; Apple 8 GB 2.6 to 3.4 or 4. The coordinator decided the cache is freed after the daily build and recommends (b) to the project lead; the public sentences hold only under (b): D3 and D4 revised |
| C9 | Index mapping option (a) multiply-shift (fades cards) against (b) power-of-two steps 2, 4, 8 GiB | spec 01 section 1.13.3, gate 1 | 8 GB and 12 GB cards | Option (b)'s 8 GiB step (about year 12) ends 8 GB and 12 GB cards on the same day; option (a) fades them one year at a time. The choice is a genesis parameter and a tier consequence nobody has put beside the options | Decision request D4 with the lifetime table of C8 | decision | decision (D4, with the card-lifetime table; the public copy now reads the step schedule as the gate 1 proposal, C31) |
| C10 | The on-die-cache recompute chip's gain is 2.4x at every scratch share under the 6 GB cap; the mixer multiplier x2 brings it to 1.2x, x4 to 0.6x, inside the 10 ms verify gate | `docs/analysis/scratch-soundness.md` finding 2 and section 10; M16 analysis table | chip builders; the site's "under 2x" claim; pool share verification | Layer 3 does not deliver the headline and the rollout plan's own rule (6a) says the public claim is qualified when no share gets the chip under 2x. The lever that does is M16's mixer multiplier, which doubles the CPU verify per warp (0.441 ms to about 0.9 ms at x2): a pool core verifies about 11,000 members' shares instead of 22,000, and node block verification doubles. The corrected SRAM cost ($19 to $37 of silicon per mirror die) means the mirror never stops a funded chip; the cache schedule keeps it above GPU L2 only | The coordinator either carries the mixer x2 into class v3 for the devnet (it is a lottery-hash change like the others; the verify cost per warp is measured with it) or marks the site's "under 2x" as qualified in the public copy level 3 and D5 asks the project lead which | coordinator ada8afb62d752b1e2 | closed as a consequence (coordinator: the public claim was qualified at 21:50; at 22:00 the M16 mixer x4 was decided into class v3 behind the same activation; the pool-core and node verification consequences are C14) |
| C11 | RX 9070 XT 17.73 MH/s at 198.9 W (0.089 MH/W) against the RTX 5090 122.30 MH/s at 307.6 W (0.398 MH/W); every read width costs the 9070 XT the same 2.4 G line fetches a second | bench-log AMD telemetry entry; readwidth probe ceilings (status 20:35) | AMD home miners; the "three vendors" copy | Under the "widest latency-bound read" rule w16 moves bytes per hash, not loads per hash, so the AMD card keeps about a seventh of the 5090's hash rate and pays 4.5x the electricity per hash; at a UK tariff of 25 p/kWh (approximate) the 9070 XT spends 4.5x the 5090's pence per IGN. The readwidth decision table carries MH/s only; no MH/W or MH per pound per card per class, and the public copy implies vendor parity | The readwidth table adds MH/W (and MH per pound at list prices, approximate) per card per class; the v3 decision names the AMD consequence; whether the class should favour fewer loads per hash for AMD's 64-byte lines is a consensus choice for the project lead (D6) | read-width a451c9935bfb1bc19; coordinator; decision | closed (read-width e752fc7 section 4.1: MH/W and MH per pound per class per card; the AMD gap is the card's random-access rate, no width closes it; the coordinator's 22:30 entry says "near parity per pound" where the table says the 5090 is 2.2x per pound: C18) |
| C12 | The v3 working set on Apple: 1,568 to 1,760 MiB today, 2,592 to 2,784 MiB at the 2 GiB genesis dataset, in unified memory shared with macOS; 2,048 launched warps x 128 KiB scratch | `docs/plans/hot-table.md` section 3 M5 Max row; era-layout section 3 | Apple silicon with 8 GB and 16 GB (the base Mac mini, MacBook Air) | An 8 GB Mac holds the miner's 2.6 GB beside macOS's 3 to 4 GB: it mines today and swaps at the first dataset growth step. The app has no gate on Metal's `recommendedMaxWorkingSetSize`; the site's "Any 4 GB card" has no Apple line. The CPU prover's RAM on a 16 GB Mac is unmeasured (the Settings switch lets it on) | The app reads `recommendedMaxWorkingSetSize` and refuses to mine (with the reason on the Mine tile) when the working set exceeds it; the site says "Mac: 8 GB or more"; the Apple scratch row at 128 KiB is measured in the readwidth table, not launched at 2,048 by assumption | miner UI adbf58b058186a18b; read-width (the Apple row) | in work (read-width 30ff674: the Apple scratch footprint by arithmetic is 1.4 GiB at 2,048 x 32 KiB and 1.6 GiB at 2,048 x 128 KiB, 2.4 GiB at 4,096 x 128 KiB; the Metal harness now prints currentAllocatedSize and recommendedMaxWorkingSetSize in its RESULT line, the measured row waits for the Mac measure lock, held by a prover measurement since 20:31Z; the miner UI gates on the arithmetic plus its output buffer until then, mining.gate_reason next cut) |
| C13 | `/api/supply` `max_supply_ign` 4,000,000,000 against `minted_at_end_ign` about 3,963,000,000 (the ramp withholds about 37 M, the floors the rest); the live tables lack `number`, `tx_count`, `detail` until the observer restarts | `site/api/supply.mjs` 31 to 48; `docs/plans/explorer.md` "Open" | everyone who reads the explorer beside the homepage tile "4B IGN hard cap, ever" (`site/index.html` 386) | The tile says 4 billion and the explorer page says "of 4,000,000,000 by the rule" while the API's own end figure is 3.963 billion: a reader who adds the two columns finds 37 million missing. The explorer page shows "Circulating 0 IGN" on the devnet deployment until the observer restarts on the new code | The tile, the litepaper's supply line and the explorer carry "cap 4,000,000,000; about 3.96 billion ever minted" (the litepaper already says "approached and never reached"); the explorer does not go live before the observer restart lands the columns | explorer a76f60b415859b7b5; wording: the project lead (D7, with D3) | closed (explorer 3e01212: the tile reads "cap 4,000,000,000, 3.96 billion ever minted, x% so far" with the withheld figure on hover; the homepage tile and litepaper wording stay with the project lead, D3 and D7; the zero-circulating premise was sharper than stated, a 42703 failure not a null, now a 503 with the reason, and moot: the live observer restarted on the explorer code at 19:42:50Z, so the live tables carry the columns) |
| C14 | The M16 mixer x4 decided into class v3 at 22:00 (coordinator): verifier 1.6 to 4.8 ms per warp against 0.441 ms today (0.87 cold), measurement running on branch ca2-mixer | coordinator's reply 21:5x; `docs/analysis/m16-recompute-attacker-2026-10-05.md` table; spec 09 section 9.8 item 5 | pool operators; every node (the Hetzner seeds, the observer, a 2019-class laptop); the 10 ms verify gate | At x4 one pool core verifies 200 to 600 shares a second instead of 2,270, so one core covers 2,000 to 6,000 members at one share per 10 s instead of 22,000; a block's CPU re-check and the miner's own `cpu re-check` of every found hash cost 4 to 11x; the seeds' small VMs verify every header at that cost; the gate's margin falls from 23x to 2 to 6x, which bounds Counter ASIC 3.0's room | The coordinator is adding the numbers to the ca2-mixer document and the status (said in reply); the reviewer checks at the next sweep that the pool-members-per-core and the seed-VM header-verify rows are there, and that the spec 09 figure 2,270 shares a second per core is re-cut with v3 | coordinator ada8afb62d752b1e2 | closed with C19 (the same measurement: x4 is 1.45x and x8 2.1x the verifier, not 4 to 11x; a pool core verifies 1,140 shares a second at x4 and 790 at x8 quiet) |
| C15 | A full prototype shard peaks at 28.3 GB on sp1-gpu-server 6.8.1 (the sweep, proving agent 22:0x); an empty one 13.9 GB; no knob moves either floor | proving agent's reply; sweep job `memsweep-pc2-pv1` | 24 GB cards (4090, 7900-class if it had a path); the 5090 that mines and proves; the fleet table | A 24 GB card cannot prove a prototype full shard at all, mining or not; a 5090 mining (3.4 GB resident) plus a full shard is 31.7 GB against 31.8 GB, the edge. The 20 GB / 16 GB default rests on the empty-shard number. From H tomorrow the fleet proves v1 shards of 30,000 pgas (about 7 M cycles, a ninth of the prototype shard) whose peak is unmeasured; `proving/fixtures/fees-v1-shards2` and `-shards3` are that shape. The fleet table's "proving-only" rows are 32 GB-card rows until then; the litepaper's "12 GB" (D2) and the evidence row 16 fall with it | The sweep's last row is a v1-budget shard with and without the miner, and the provedefault gates are set from it before 0.3.11 ships default-on; the fleet table labels its rows by the card that fits | proving v1 | closed as measured (proving agent, the S_p curve, app default 440fd59): 32 GB mines and proves today (28.3 GB alone, 30.0 beside the miner); 24 GB proves the adopted 30,000-pgas shard (20.4 GB alone, about 22 GB beside the miner) from DAA 210,000 and nothing before it; 16 GB proves only empty shards alone (13.9 GB), nothing beside the miner (15.7 GB); 12 GB proves nothing on SP1 6.8.1; the default is on at 24 GB or more. Relayed to the rig installer and the HiveOS words sub-agent for their tables; until H tomorrow every prover on the devnet is a 32 GB card |
| C16 | PC 2's app quit at 20:01:09Z ("aborted (the app is quitting)"), logged by the proving agent as "the app quit for the 0.3.10 update"; the shipper says nothing of 0.3.10 was published and no update-now of its exists (the live manifest is 0.3.9 from 17:59:14Z) | bench-log "proving v1" chain run 1; the shipper's reply 22:0x | every measurement on PC 2 that straddles 20:01Z (the prover-cost phase B ended 19:51Z, the chain re-run began 20:05Z, the readwidth 5090 job queued) | An app that quits for an unknown reason voids any number taken across it (CLAUDE.md: a number taken while another build or simulation ran is not a number; the same for a restart). The bench-log line names a cause that did not happen | The proving agent reads PC 2's app log for the quit reason at 20:01Z and corrects the bench-log line; if the cause is another agent's job or the auto-update, that agent's measurements across it are marked | proving v1 (the log read); the agent the cause names | closed in part (proving agent: PC 2 logged "quit: stopping the miners, then the node" at 20:01:09Z, a plain quit command 20 s after the efficiency sweep's administrator prompt was cancelled at the keyboard and 13 s after the live prover failed on a root-owned /tmp/sp1-cuda-0.sock left by the chain job; the bench-log line corrected; the quit's origin is not in the log, see C17) |
| C17 | The chain job ran igneum-prove-host as root inside WSL2 and left a root-owned `/tmp/sp1-cuda-0.sock`; the live prover (the app's user) then failed with `CudaClientError: Connect(PermissionDenied)` at 20:00:56Z; the app quit at 20:01:09Z on a command whose source the log does not name | proving agent's reply 22:1x; PC 2's app log | every PC 2 measurement that shares the card with the live prover; every operator whose machine takes remote jobs | Two classes, not one bug. (1) Any job script that runs the SP1 server or the host as root in WSL2 breaks the live prover for every later shard until a reboot or a manual unlink; the proving agent fixed its own scripts (kill the server, remove the socket at the end), the class check (CLAUDE.md, 5 October: grep every script with the same shape, add a check that fails when the shape comes back) is not yet written. (2) A quit the log cannot attribute (job, UI, signal, update) voids the measurements around it and nobody can say who stopped a miner; the app logs "quit:" without a source | (1) `tools/ci/` gets a check that fails on any `.ps1` or `.sh` playbook that invokes `igneum-prove-host`, `sp1-gpu-server` or `wsl -u root` without the socket cleanup line, and the proving agent greps tonight's four PC 2 scripts; (2) the engine's quit log line carries its source (job id and playbook name, the UI, a signal, the updater) in the next app cut | proving v1 (1); coordinator for the next-cut list (2) | closed in part (proving v1 c2544be: every pv1 playbook kills the server and unlinks the socket at start and end, tools/ci/prover-socket-check.sh in CI; the publish-time gate is C27 on bash-body-check 6805125; the unattributed quit, the engine logging its source, is on the coordinator's next-cut list) |
## Round 2 (22:30 to 23:30)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C18 | 5090 0.072 against 9070 XT 0.032 MH/s per pound at list prices (2.2x), 4.9x per watt | read-width.md 4.1 | AMD home miners; the public level 3 copy | The coordinator's 22:30 status entry says "near parity per pound"; the table it cites says 2.2x. The level 3 page is written from the status | The status and level 3 carry the table's figure (2.2x per pound, 4.9x per watt, 7.5x in rate); "near parity" is struck | coordinator | closed (coordinator 6b07776: the status, the rollout plan and the public copy carry 2.2x per pound, 4.9x per watt, 7.5x in rate) |
| C19 | Mixer x4 into class v3: CPU verify 1.6 to 4.8 ms per warp against 0.604 today; pruning depth 108,000 DAA s (spec 02) | status 22:00; M16 table; spec 02 line 17 | every node (the three testnet seeds on small Hetzner VMs, the observer, a laptop node); IBD; pools | A new node verifies every header in the pruning window on one core: 108,000 x 4.8 ms = 8.6 min at x4 against 1.1 min today (a 30-s block-time budget at 1 block/s is unaffected: 4.8 ms per block is 0.5% of a core). Every miner's own `cpu re-check` of a found hash and every pool share verification cost the same 8x; the 10 ms gate (ledger M9) keeps 2x of margin at the slow end, which is what Counter ASIC 3.0 has left to spend. On the 2019-class laptop core the evidence table names (rule 3) the figure is unmeasured and may pass 10 ms | The ca2-mixer document carries: ms per warp on the M5 Max core AND a scaled 2019-class figure (marked approximate), the pruning-window IBD minutes per tier, pool shares per core per second, and the gate margin left for 3.0; the seeds' header-verify load is checked in the testnet go checklist | ca2-mixer af345b1e2c541ffbb; coordinator | closed as measured (ca2-mixer 54bbfcc, mixer-x4.md 6.5): one M5 Max core under a load of 5.6, ms per warp v2 1.33, x4 1.94 (1.45x), x8 2.79 (2.1x), worst cold 1.58 / 2.04 / 2.94; quiet-core scaled 0.60 / 0.88 / 1.26 (approximate); 2019-class laptop 1.5 / 2.2 / 3.2 (approximate, unmeasured); shares per core per second quiet 1,660 / 1,140 / 790 (spec 09's 2,270 re-cut to 1,660), a 22,000-member pool needs 1.3 / 1.9 / 2.8 quiet cores; IBD over 108,000 headers quiet 1.1 / 1.6 / 2.3 min (laptop 2.7 / 4.0 / 5.8); gate margin for 3.0 on the loaded core 8.4 / 8.0 / 7.1 ms. The mixer multiplies the ALU part only, so x8 is 2.1x the verifier. Owed before the level 3 page quotes an absolute: one quiet-core run (the ratios are the measurement tonight). Superseded at 22:07 by the fixed crate: v3 (x8) 2.1 ms per warp near-quiet, 3.4x v2 (see C29) |
| C20 | Layer 9: the epoch length as an era parameter, 600 to 7,200 DAA s (10 min to 2 h), base 3,600 | status 22:30 and 23:00 | rigs (HiveOS and the rig installer), Macs, pools, the seed path | Both rig miners run `--exit-on-seed-change` and re-export the pack on exit 42 (h-run.sh 56 to 61, igneum-miner.sh 67 to 78): at a 10-minute epoch every card's miner restarts six times an hour with a pack export each time, and the restart gap is lost hashing; the app's prepare-ahead path does not restart. The Mac fleet's prepare pause (35 s an hour at one epoch an hour, bench-log M11) becomes 3.5 min an hour, 6%. The 10-minute seed VDF (spec 04) equals the shortest epoch, so the seed for epoch n+1 is known only as epoch n starts, which is the compile-ahead window the agent must measure per card (the 5090 compiled in 1,285 ms, the 9070 XT unmeasured). A pool's `job` cadence and the dev-fee counter are unaffected | The epoch-length document carries a per-tier row: rig restart cost per epoch length (and the fix: prepare-ahead in the rig scripts, no exit 42 path), the Mac pause share, the compile-ahead margin per card at 600 DAA s against the VDF; the rig installer removes `--exit-on-seed-change` in favour of the prepare path before layer 9 can draw a short epoch | ca2-epoch a32a3ece66c02417a; rig installer | closed with a correction (ca2-epoch 4300608, epoch-length.md sections 6.3, 7, 9): the rig scripts pass --prepare-packs as well, so a worker with prepare support swaps in place and exit 42 is the fallback on a prepare MISS (the loaded iGPU missed 2 of 10, M11), not a restart every epoch; the risk at 600 s is one restart plus export plus inline compile per missed boundary; the Mac race pause is 6.3% of a 600-s epoch (race default off); the program is known a full epoch ahead at every length (lead and T_epoch fixed, option A); the 9070 XT compile is OWED (no prepared line from gfx1201 in any upload); the rig installer (88f31d9) confirms the rig already takes the prepare path: exit 42 fires only when a worker's ready line lacks "prepare 1", and both shipped workers answer it; documented in its README, no code change |
| C21 | OTA K2: apps embed `OTA_PUBLIC_KEYS = [K1, K2]` and honour a signed `revoked_keys` list; the rig installer verifies the manifest with ONE key (`OTA_PUBLIC_KEY_HEX`, install-rig.sh 168, igneum-update.sh 2) and knows no revocation; the HiveOS package verifies nothing (no manifest, the override reaches it only by republish) | ota-k2 c722579; packaging/linux; packaging/hive | rigs (both packages), the seeds (if they take the manifest) | The day K1 is lost or revoked and the manifest is signed with K2, every rig on the installer refuses the manifest, stops taking overrides, and is isolated at the next height switch; a leaked K1 keeps signing for rigs, because they carry no revocation list. HiveOS rigs get neither keys nor revocation: a republished package is their only path, and nothing checks who published it | The rig installer carries both public keys and the `revoked_keys` rule in the same form as the app (keys.md section 4, step 3), installed and read from the manifest; the HiveOS README states that the package is unsigned and names the sha256 the Flight Sheet URL should be checked against; keys.md lists the rig and HiveOS paths in its table of what trusts K1 | OTA key af2bb75a5436324d0 (keys.md, the shared verifier form); rig installer a3e7b2b03222f5cff | closed for the rig and the docs (rig installer 88f31d9: OTA_PUBLIC_KEYS [K1, K2 slot] embedded, manifest_check mirrors the app's manifest::check with the revoked_keys record at /var/lib/igneum/updates/revoked.json, tested on three throwaway keys; OTA agent 00fcbb5: keys.md table of every path that trusts K1, the HiveOS README unsigned-archive note); open: the wallet (wallet-v1) still trusts K1 alone, listed in keys.md for its owner; merge note: packaging/hive/README.md is edited on both ota-k2 and hive-words |
Sweep 3 (20:46 Mac clock) notes, no new row: the 9070 XT dropped off PC 1's bus at about 20:40 UTC (the second eGPU fault of the day); the coordinator stated the consequences (G1 on the gfx1036 stand-in, the 9070 XT v3 hash-rate and power rows owed, every earlier 9070 XT row stands, the AMD sweep queue item blocked, nobody woken) in its 21:05 and 21:10 entries and the rollout plan 7b. The epoch-length plan (ca2-epoch 4300608) carries its own per-tier table (section 7), including the node tier (one core 100% busy on the VDF at the 600-s floor) and the chain (24% of blocks in difficulty settle at the floor). The proving agent's 440fd59 rewrote the litepaper's two proving-gate sentences and evidence rows 15 and 16 on its branch (D2: the project lead approves the draft); the litepaper's card-lifetime sentence is untouched (D3, D4).
| C22 | The SP1 CPU prover peaks at 29.5 to 30.5 GB RSS whatever the shard size and costs 282 s a shard (PC 1, `cpu-prove-pc1-small2`); no zkVM proves on AMD; the analysis concludes "no CPU tier" | `docs/analysis/amd-proving.md` sections 2, 3, 4a (amd-prove f1d7a7d, merged into ca2-coord) | Macs with 16 or 24 GB (the Settings switch turns the CPU prover on); AMD-only Windows and Linux machines (Settings can switch it on); every tier's expectation of the 20% share | provedefault.rs has a RAM gate for Windows (31,000 MB) and none for macOS or Linux, so a 16 GB Mac that flips the switch runs a 30 GB prover into swap and takes the node down with it (the Mac went down at 1% battery on 4 October; this is the same class of outage from memory). The analysis says the tile must say "about five minutes, paid only when no card proves first" but not that the switch is refused under 32 GB. The public tiers: AMD and Apple miners never see the 20% share on this build (now on the site, 1c8439f) | The CPU-prover switch is refused with the reason on every OS under 32 GB of RAM (the Windows constant generalised: macOS reads hw.memsize, Linux /proc/meminfo), and the tile line carries the 5-minute and 30 GB figures | proving v1 acd4f36bc2c07a4e2 | taken in full (proving agent: the prover loop refuses the CPU path on every OS under 32 GB, Settings cannot bypass it, with the line naming the machine's RAM; the tile line on CPU machines carries the 5-minute, 30 GB, paid-only-if-no-card figures; lands in the next app commit after the gate-test build; the "measured on a 32 GB card, not yet on a 24 GB card" marker is in evidence row 16, both litepaper sentences and the plan) |
Sweep 5 (21:08 Mac clock) notes: the coordinator applied the public proving line on ca2-coord (1c8439f: index, litepaper, miner page: "an NVIDIA card with 24 GB or more proves; AMD and Apple cards mine; a prover for them lands when a zkVM ships one") and the proving agent rewrote the litepaper's two gate sentences on proving-v1 (440fd59): two drafts of overlapping public sentences on two unpushed branches, both for the project lead (D2, D8); the integrator takes one. The measure-lock convoy (a0c3d13: a dead holder, two waiters, cargo tests re-acquiring build slots) is stated by the coordinator with a next-cut task. The S_p CPU shard job was dropped on the 312-s small-shard number (stated). GitHub Actions outage: the 0.3.10 installer builds on PC 1 (stated, 21:01).
## Round 3 (21:30 to 22:00 Mac clock)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C23 | Mixer x4 or x8: the 5090's daily dataset build 13.4 ms at x1 (54 ms at x4, about 107 ms at x8); the integrated gfx1036 builds the dataset per PREPARE, 7 to 12 s at x1 and 55 to 124 s under CPU load (epoch-length.md section 7); the x8 rule: "the daily build under 1 s on every card we own" | status 21:25 and 21:27; epoch-length.md section 7 | integrated GPUs (the iGPU tier), 8 GB cards (about a tenth of a 5090's rate), rigs with one weak card | The x8 rule names "every card we own": the gfx1036 is one, and at x8 its per-prepare build is 56 to 96 s per epoch (7 to 16 min under load), so it misses every epoch boundary and falls to the exit-42 path; at x4 it is 28 to 48 s per hour (0.8 to 1.3%) or 4 to 8 min under load. An 8 GB discrete card scales at about a tenth of the 5090: about 1 s at x8, on the rule's edge. The per-day dataset reuse in the worker (owed in epoch-length.md section 9) is what makes the mixer cheap for the iGPU tier; without it the mixer multiplies a per-epoch cost | The mixer decision names the gfx1036 and an 8 GB-class scaled row beside the 5090, M5 Max and 9070 XT in the "under 1 s" check, and the per-day dataset reuse in the worker lands before (or with) class v3, or the iGPU tier is stated as "mines v3 with a restart per epoch" on the level 3 page | ca2-mixer af345b1e2c541ffbb; coordinator | taken (coordinator and ca2-mixer, mixer-x4.md build-time table: the x8 table carries the gfx1036 row and a scaled 8 GB-class row; the under-1-s rule applies to the discrete cards' daily build; for the integrated tier either the per-day dataset reuse in the three workers lands with class v3, asked of the mixer and node agents as a bounded change tonight, or the level 3 page says the iGPU tier mines v3 with a restart per epoch; recorded with the x4/x8 choice) |
| C24 | Two PowerShell job scripts lost a quote inside an inline bash body tonight: amd-prove's awk program (cpu-prove-pc1-small, exit 0 with nothing proved) and the 0.3.10 installer's `bash -c` string (fb-installer-pc1-3, exit 2 in 4 s); amd-prove added `tools/amd-prove/check-job-bash.sh` (bash -n on its own scripts' bash bodies) | amd-proving.md section 2; status 21:28 | every PC job; the morning's rollouts | The class (CLAUDE.md, 5 October: fix the class the same day, add a check that fails when the shape comes back) is "a bash body inside a PowerShell job string"; the check exists for one agent's scripts and did not cover the shipper's, which failed the same way an hour later | A repo-wide CI check: every `.ps1` under relay/playbooks and tools that carries a bash body (`wsl ... bash -c`, `bash -lc`, here-strings fed to bash) has that body extracted and passed through `bash -n`; the shipper's rule (the WSL part as a file run with `bash <file>`) written in packaging/README-ship.md as the convention | sub-agent (bounded, no owner); coordinator told | closed on branch bash-body-check 7adb1ca (tools/ci/bash-body-check.sh with fixtures and self-test, ci.yml, the convention in packaging/README-ship.md; the flagged existing playbooks are in the sub-agent's report for their owners) |
| C25 | Ember Tune: the signed manifest carries per-card-model tuning priors (power limit and core clock) that a new card applies and confirms in two steps; K1 signs it | ember-tune.md sections 4 and 5; bench-log Ember Tune entry | every NVIDIA and AMD card on the app; the keys | The manifest now sets clocks and power limits on every user's card, so the signing key's blast radius grew: a signed prior can underclock the fleet or push a card model to its power ceiling. The plan's clamps (inside power.min_limit / max_limit and clocks.max.gr, the confirm step, a faulted step reverted) are the bound; keys.md's "what K1 signs" table (ota-k2 00fcbb5) predates the priors and does not list them | ember-tune.md section 5 states the bound in one line (a prior can never set a value outside the card's own reported limits, and never a memory clock), with the test that proves it; keys.md's table gains the tuning priors under K1 with that bound | Ember Tune a855dcc4bd05e0615; OTA key agent (the table row) | closed (Ember Tune 5d7ced9: the rule as a row in section 5 with the two tests named, a prior of 9,000 MHz at 30% clamps to 3,090 MHz at 50%, a bad prior costs one confirm step per card; the OTA agent has the keys.md note: tuning priors and the kill switch under K1 with that bound) |
Sweep 7 (21:49 UTC) notes, no new row: the hot table is measured and NOT adopted (g 0.93 to 0.96 on the Mac against the 0.97 rule, bdab8df); the era draw passes the 5% rule on the Mac (spread 0.8%, c570da3) and its chip line says the union of a program's 16 windows covered the whole dataset in 300 of 300 programs, so a chip mirrors the whole dataset or nothing; the mixer x8 passes the verifier half of its rule (C19) and the PC build rows decide the other half about 22:10; PC 2's miners have been off since a job's /api/resume at 21:25 answered ok without restarting them (the devnet short PC 2's rate, the aggregation-cost mining phases void, a next-cut defect: a resume re-checks the miner processes), all stated by the coordinator at 21:45; PC 1 restarted on 0.3.10 at 21:40:41Z with no measurement straddling it.
## Round 4 (22:00 to 22:30 UTC)
| # | Number | Source | Tier affected | Consequence | Action | Owner | State |
|---|---|---|---|---|---|---|---|
| C26 | The 13.9 GB floor's cause: the shipped sp1-gpu-server 6.8.1 panics on any card under 24 GB (builder.rs 35 to 39) and allocates its core, recursion, shrink and wrap provers at Setup; the prover-floor agent rebuilds it from source on PC 2 with those sizes cut, CUDA_ARCHS=120 | proving-v1.md 4c82e56; status 22:01 | 12 and 16 GB NVIDIA cards (RTX 3060, 4070, 5070, 5080, 4060 Ti 16 GB): the tier the project lead asked for; packaging and signing | A server built for CUDA_ARCHS=120 runs on the 5090 only; the 12 GB tier is sm_86 (3060) and sm_89 (4070), the 16 GB tier sm_89 and sm_120, so a cut-size server measured on the 5090 proves nothing about a 3060 until the on-order 3060 runs it, and the build must list sm_86, sm_89, sm_120 (sm_100 is datacentre) to serve the tier at all. Shipping our own 250 MB CUDA server means the project signs and distributes a build of someone else's prover: it enters the DMG and the WSL2 package, the K1-signed inputs, the SBOM-style notes in evidence.md, and every SP1 upgrade is re-done by hand. Cut buffer sizes do not change the verifying key (prover-side chunking), so no guest re-pin, but the recipe must say so with a verify-segment run on a proof from the rebuilt server | The prover-floor measurement states its arch list and the card it ran on; the 12 GB claim waits for the 3060; the rebuilt server's packaging path (who builds, who signs, where it lands) is a row in the proving plan before 0.3.12, and the public line keeps "24 GB" until the 3060 proves on it | prover-floor agent (through the coordinator); proving v1 | taken (the proving plan carries "A self-built CUDA server (the 12 GB path), before 0.3.12": arch list sm_86 / sm_89 / sm_120 with one measured row per family, the build on PC 1 from a pinned SP1 tag, the Mac signs, placement as wsl2/bin/sp1-gpu-server with its sha256 in payload-inputs.json and the DMG, the evidence.md note, the rebuild at each SP1 upgrade, the gate that verify-segment and verify show the pinned keys unchanged; the 12 GB claim waits for the 3060; the prover-floor agent abefda4c3872f866f has the measurement side) |
| C27 | The root-socket fault recurred at 21:25Z from another agent's job (agg-cost-pc2-1) after the class fix and CI check landed; PC 2's prover was dark 37 minutes; a resume at 21:25 answered ok without restarting the miners | bench-log 1d78979; status 21:45, 22:03 | every PC job; the devnet's proving and hash rate tonight | The class check lives in CI, but PC jobs are published from worktrees by `publish-jobs.sh` and never pass through CI before they run, so a job written on a branch without the check runs the old shape. The check must run where the job is published, not only where the repo is tested | `packaging/ota/publish-jobs.sh add` runs `tools/ci/prover-socket-check.sh` and the bash-body check on the script it publishes and refuses on a failure; the sub-agent on the bash-body check wires both; the resume defect is on the next-cut list (stated) | sub-agent bash-body-check (the wiring); coordinator (the rule) | closed on branch bash-body-check 6805125 (publish-jobs.sh add --kind run runs the bash-body check and prover-socket-check.sh before signing, refuses with the output, never skips; test-publish-jobs.sh 32 passed with four new refusals). Merge notes for the integrator: prover-socket-check.sh exists on both proving-v1 c2544be and this branch (add/add, take the superset here); the socket grep flags tools/amd-prove/pc1-cpu-prove.ps1 (CPU-only, -u root, no GPU server) so that job needs the cleanup lines or an allow-list entry before its next publish, told to the coordinator |
Sweep 8 (22:09 UTC) notes: mixer x8 DECIDED into v3 on the PC rows (the daily 1 GiB build latency-bound on every card: 5090 23 to 25 ms, 9070 XT 72 to 77 ms at every multiplier; the chip row 0.92x with the 3x factor), so the public claim holds with margin (D5 re-cut); the verifier regression (2.2x) bisected to inlining in the mixer's fetch loop and fixed, so the quiet-core figures of C19 return to about 0.6 / 0.9 / 1.3 ms; G6 job 3 failed on a stale fork test (era inside the class), job 4 on the final tree; the integration merge into master has three known conflicts (bench-log append-only, packfile.h and host.c take the ca2-v3 side).
| C28 | One sp1-gpu-server per card on rigs pins about 6 GB of host RAM per server today (under 2 GB with route A's buffers, approximate) | `docs/analysis/proving-methods.md` section 5, rig row (proving-methods e7e0db7) | rigs with several 24 GB or 32 GB cards on the rig installer | A six-card rig that proves on every card pins about 36 GB of host RAM under the shipped server; the rig installer's preflight says 16 GB to prove (a warning, per card not per rig) and its prover unit runs one server, so the moment it moves to one server per card (the analysis's route C) the RAM check is wrong by the card count | The rig preflight scales its RAM warning by the number of proving cards (6 GB each today, the route A figure when measured) and the README's per-card table gains a host RAM column; the one-server-per-card unit lands only with that check | rig installer a3e7b2b03222f5cff | closed (rig installer 086008a: the preflight checks host RAM against 8 GB plus about 6 GB per proving card at the 23,552 MB gate, PROVER_RAM_GB_PER_CARD default 6 marked approximate, the README host RAM row per tier; the one-server-per-card unit lands only with that check) |
Sweep 8a (22:2x UTC) note: `docs/analysis/proving-methods.md` (branch proving-methods e7e0db7, 411 lines) carries its own per-tier table for today, route A, route D and route 3, and a public paragraph; it is the third draft of the proving public line (with proving-v1 440fd59 and ca2-coord 1c8439f), noted under D2 and D8; the route choice is D10.
| C29 | The litepaper's verifier line now reads "Measured 2.1 ms on one loaded Apple M5 Max core for class v3" and the status says "4.8x inside the gate"; the measurement (ca2-mixer 54bbfcc) is 2.79 ms per warp for x8 on the loaded core (2.1 was the RATIO to v2), worst cold unit 2.94 ms, quiet-core about 1.3 ms by scaling | site/litepaper.html on ca2-coord 8e65696; status 22:20 | the public page; every node operator who reads the gate margin | A ratio printed as milliseconds understates the verifier cost by a third and overstates the gate margin (10 / 2.79 = 3.6x, 3.4x on the worst cold unit, not 4.8x). The number is the one a reviewer will re-run first | The line reads "about 2.8 ms per warp on a loaded M5 Max core (about 1.3 ms quiet, approximate), 2.1x the v2 verifier; worst cold unit 2.9 ms; the 10 ms gate leaves 3.4x" until the quiet-core run lands | coordinator ada8afb62d752b1e2 | closed, the reviewer's reading WITHDRAWN in part (coordinator 7e6f77c, status 22:25): the fixed crate's session at 22:07 measured 2.1 ms per warp for class v3 as a MEASUREMENT (worst cold 2.15) on a core at load 5.5, and that binary's v2 figure matched readwidth's quiet 0.61 ms within 1%, so the numbers are near-quiet and the litepaper's 2.1 ms was right; the 2.79 ms I cited was the slow binary's 21:40 session (the inlining regression, since fixed); the gate leaves 4.8x (4.6x on the worst cold unit), 3.4x the v2 verifier. The "about 1.3 ms quiet" scaling is struck everywhere. C19's quiet-core figures are superseded by this session |
| C30 | The 5090 mines in the app at 115.4 MH/s (the power sweep, 22:09 to 22:15Z, hash from the app's API) against 136 to 137 MH/s at device time in every bench row tonight, with the cap not binding (draw 316 W under a 431 W cap, SM at 3,051 MHz) | bench-log e304458; read-width and mixer PC rows | every 5090 owner on the app (and every big card: the gap is the app's job loop, not the kernel) | About 15% of a 5090's hash is lost between the kernel and the app, and the sweep entry explains it away as API sampling. M11 measured the 9070 XT at the app's 2^21 job size equal to its 2^24 rate, but no 5090 row exists at 2^21; the 5090 finishes a 2^21 job in about 15 ms, so per-job launch, read-back and template work can cost that much. The STATUS line prints "wall" and "inside jobs" rates and would show it | One measurement on PC 1: `igneum-worker-cuda --bench` on the live pack at --batch-log2 21 and 24 on the 5090, and the 5090's STATUS wall-against-inside gap over 10 minutes; if the job size is the cause, the app's job size for cards over 100 MH/s rises (2^22 or 2^23) in the next cut: a 15% gain for every 5090 owner | repro-bench agent a0b9f574775ef1693 (its PC 1 slot); coordinator | taken (repro-bench agent: --bench at 2^21 and 2^24 on the 5090 on the genesis pack and the live pack in its PC 1 slot, then PC 2; the STATUS wall-against-inside gap from the 10 minutes before and after its window; the consequence written either way) |
Sweep 9 (22:2x UTC) notes: the devnet at 22:24:31Z reads DAA 136,578, 0.909 blocks/s measured over the stats window and 1.005 DAA/s averaged since the fee-switch plan's 15:40Z read (112,227), so H = 210,000 lands between about 18:50 and 19:35 UTC on 6 October, up to an hour EARLIER than the 19:50Z written in fee-switch-devnet.md, the rollout plan 7a and release-0.3.10.md; the 16:00Z check (D1) keeps about 2.8 hours of margin and stands; the plans' ETA is re-cut in D1 and sent to the coordinator. Gates tonight: G3, G4, G6 green on the final class (x8 + era); G4b added (the Mac app passes --prepare-packs only to non-Metal workers, so every Mac would stop at the first v3 epoch: found by the node agent, the fix with a unit test and a real Metal gate run before the ship); G1 and G2 on the PC 1 job since 22:16.
Merge note for the integrator (22:3x UTC): branch `consequences` is docs/plans/consequences-2026-10-05.md and consequences-decisions.md only (base ca8d9f3); a merge-tree against master 1f0d62c shows 0 conflicts. Branch `bash-body-check` (7adb1ca, 6805125, e3bd761) carries tools/ci/bash-body-check.sh, kit-path-check.sh, the copied prover-socket-check.sh (add/add with proving-v1's: take bash-body-check's), ci.yml steps, the publish-jobs.sh gate and packaging/README-ship.md; it is on the coordinator's ship order after ca2-coord.
| C31 | The copy the ship takes (ca2-coord at 22:3x): litepaper line 562 "12 GB or more proves full shards" beside line 452 "Proving needs an NVIDIA card with 24 GB or more"; evidence row 16 still "A 12 GB card proves one shard in about 20 s, designed" while proving-v1 440fd59 withdrew it; line 562 states the dataset "doubles on a step schedule fixed at genesis (years 4, 12 and 28)" | ca2-coord site/litepaper.html 452 and 562, docs/evidence.md 43; proving-v1 440fd59; D4 | every reader of the litepaper; the integrator; the project lead's D4 | One page says 12 GB and 24 GB for the same thing; the evidence table on the ship branch contradicts the measurement, and the two branches will conflict on evidence.md and litepaper.html at the merge (ship order: ca2-coord before proving-v1), so the stale row can win by accident; and the step schedule is written as a genesis fact while the coordinator's own 22:50 entry calls mapping (b) a recommendation for the project lead (gate 1, D4): a public page should not decide a genesis parameter before he does | On ca2-coord: strike "12 GB or more proves full shards" from line 562 (line 452 is the sentence); take proving-v1's evidence row 16 (WITHDRAWN, 24 GB measured) at the merge and say so in the merge plan; write the growth sentence as the recommendation it is ("the plan is a step schedule ... decided at gate 1") until D4 is taken | coordinator ada8afb62d752b1e2 | closed on ca2-coord a7be43f and 0d9b23d (the 12 GB clause struck; the merge rule "take proving-v1's row 16 and its proving sentences" in the rollout plan; one wording in all five places, "the proposed schedule, fixed at the testnet genesis: 2 GB, doubling at years 4, 12 and 28", the lifetime sentences kept as consequences of the proposal and marked approximate) |
| C32 | The 0.3.10 install at 21:49Z cleared PC 1's app jobs folder and with it the AMD kit fetched at 21:23:59Z; the amd-card-test playbook now says its fetch must be republished after any app update | bench-log 39f02ff; status 21:05 ("the jobs folder is cleared by fetch jobs" was the earlier, wrong reading) | every PC job tonight and tomorrow; the 0.3.11 rollout | A class, not one playbook: every fetch-then-run pair (era, hot table, mixer, repro, Ember, the AMD sweep, the prover-floor build) loses its kit when an update lands between the fetch and the run, and the run fails in seconds or, worse, runs against a stale copy. The 0.3.11 update-now reaches PC 2 while the prover-floor agent's 60 to 90 minute server build runs there (go at 22:17, to about 23:50): if that build's working directory is under the app's jobs folder, the update wipes it mid-build and the 12 GB rows slip past the morning | (1) The 0.3.11 update-now is sequenced after the prover-floor build closes, or the build's directory is confirmed outside the jobs folder before the ship; (2) every run playbook begins with a presence check of its kit and fails with "kit missing: republish the fetch after the app update" (the class check: the bash-body sub-agent's CI check gains a rule that a run job naming a kit path tests it first, or the coordinator's queue re-fetches after every update as a rule) | coordinator ada8afb62d752b1e2 (the queue and the ship order) | taken (coordinator: the prover-floor build lives under /opt/igneum-floor in WSL2, outside the jobs folder, but it is the app's job process and an app restart ends it, so the 0.3.11 update-now goes to PC 2 only after floor-build-3 closes, the ship's earlier steps not waiting; the re-fetch rule and the presence-check rule are in the rollout plan beside the one-job rule, the playbook owners carry it at their next publish; the CI side is with the bash-body sub-agent as a kit-path check). CI side closed on bash-body-check e3bd761: tools/ci/kit-path-check.sh in ci.yml and in publish-jobs.sh add --kind run (34 tests pass); every existing kit-using playbook (13 across master, ca2-v3, ca2-analysis, rdna4-telemetry) already checks before use, so the gate guards the shape without a backlog |
| C33 | `/api/live` at 22:33Z: 13 of 482 blocks fully proven in 10 minutes (2.7%), 1 prover, median proof lag 46 s, the live node's verifier "Off" with 42 pool entries pending and 0 verified; `/api/stats` (the documented public API) carries no proving field at all | live and stats handlers (`site/api/stats.mjs` FIELDS; the explorer branch 3e01212); the homepage "~60 s to a proof"; evidence row 15 | everyone who reads the public API or the homepage tile; the testnet's first external reader | The public stats API hides the one number that qualifies the tile and row 15: coverage is 2.7% with one prover, and the proof lag is 46 s. A reader can find it only on /api/live. The live node's verifier "Off" (42 pending, 0 verified) is the Mac app node in trust mode or without a host, so the page says "verifier Off" while the chain pays provers: a public-page oddity the morning reader will ask about | `/api/stats` gains a `proving` object from the same live_state (`blocks_10m`, `blocks_fully_proven_10m`, `shards_paid_10m`, `provers_10m`, `median_proof_lag_s`, `active`), documented in docs/api/public-stats.md and in its contract test; the live page's verifier line names which node it reads and why it is off; the homepage tile's "~60 s" caption cites the measured 46 s median and the 2.7% coverage ("the target; today one prover covers 2.7% of blocks at a 46 s median") | explorer a76f60b415859b7b5 (the API); the tile caption: D2 wording for the project lead | closed on explorer d7e797c (/api/stats carries the proving object with coverage_10m, 0.0273 at 22:33Z, in the contract test, the live check and docs/api/public-stats.md with the sentence that coverage is what a third party reads before "every block is proven"; the live page's proving legend and the API note say the observer's node runs its own verifier off and reads paid shards from the chain). The homepage tile caption stays with the project lead (D2) |
Sweep 11 (22:38 UTC) notes, no new row: gate G4b GREEN (a real Metal miner across a v3 boundary; a second Metal-only fault found and fixed first, 00c55aa: serveDataset keyed the day dataset by day alone and would have hashed v3 over an x1 dataset; the coordinator's next-cut rule: every worker path mines across a boundary in the gate network before a class change ships). The reviewer checked the PCs' side of that class: the CUDA worker and the OpenCL host build each resident pair (program, cache, dataset) from the pack's own memhard.h and key it by (epoch, day, class, era) through pairIsClass (proto-cuda/nvrtc/worker.cpp 451, proto-opencl/host.c 1106 on ca2-v3 fa3c932), so the fault does not reach the PCs at N4. The patched sp1-gpu-server built green on PC 2 at 22:32:29Z with sm_86, sm_89, sm_120 (C26's arch list), the floor sweep (9 points) running; the integration merges (readwidth 30ff674, origin/master 1f0d62c) on ca2-v3 49c7e78 with every check green. All gates but G5 (the ship's build) are green; the ship waits on the merged tip.
| C34 | The rollout plan's packaged override line (counter-asic-2-rollout.md 26) carries five switches: difficulty_v2 33000, proving_v0 84100, fees_v1 210000, finality_v3 135200, program_class_v3 N4; section 8a says 0.3.11 publishes ONE object with both activations set | counter-asic-2-rollout.md 26 and 8a; proving-v1.md (the four v1 fields enter the digest only once proving_v1_activation_daa is set) | every node and every prover on the devnet at the 0.3.11 publish | If the publisher copies line 26, proving v1 ships in the binary and never activates: proving_v1_activation_daa stays at never on every node, the digest is the five-field one, the segment records are never carried, and C1's fix (the export fields) still works but the aggregator, the chain rule and the unproven rule stay off while the plan and the public copy say they are live. The four fields (activation_daa H1 = tip + 14,400 at publish, segment_blocks 8, unproven_daa 600, aggregator_share_bps 1000) must be in the packaged line, the manifest object, the hand nodes' and the seed's files, verbatim, and the expected digest read on a scratch node with all nine fields | Line 26 and the ship step name the nine-field object with N4 and H1 both set at publish and the scratch-node digest read over that object (the 22:xx "expected 0.3.11 digest with the two new fields at never" is the rolling-upgrade digest, not the activation one; both are recorded) | coordinator ada8afb62d752b1e2 (the ship runbook) | sent |

View file

@ -0,0 +1,32 @@
# Consequence decisions for the project lead (night of 5 October 2026)
Sibling of `ledger-decisions.md`. Each is a consequence of a measured number that needs the project lead: money, a public claim, or a consensus parameter outside tonight's delegation. The row number points at `consequences-2026-10-05.md`. Recommendation first, then the options.
## The eleven in one glance (what the project lead does, in the order they bite)
| # | By when | One line | What the project lead does |
|---|---|---|---|
| D1 | 16:00 UTC, 6 October | The fee switch at H = 210,000 (about 18:45 UTC by the devnet's DAA rate at 22:33Z, 1.002 DAA/s since 15:40Z; an hour earlier than the plans' 19:50) darkens every 0.3.10 prover; 0.3.11 must be on every prover first, else H moves | Nothing if the morning check passes; the coordinator holds it. Know it exists |
| D9 | this week | No 12 GB card exists here; every 12 GB number is scaled from the 5090 | Confirm whether the 3060 and 5060 Ti 16 GB that evidence.md says are on order are real; if not, buy one 12 GB card (about £250 to £400) |
| D2 | before the next site push | The litepaper's "12 GB proves" is false on this build; three drafts of the replacement exist (proving-v1 440fd59, ca2-coord 1c8439f, proving-methods.md section 5) | Pick one; the proving-methods paragraph is recommended |
| D8 | the same push | "The card mines and proves" and "a visible 1% software fee" are solo-NVIDIA sentences now | Approve the qualified wording |
| D3, D7 | the same push | "Any 4 GB card", "4 GB about four years, 8 GB more than a decade", "4B hard cap" against the lifetime table and the 3.96 billion ever minted | Approve the re-cut sentences |
| D4 | gate 1 (before the testnet genesis) | Dataset growth mapping: (b) power-of-two steps (years 4, 12, 28, 60) keeps the public sentences true; (a) fades cards one year at a time | Choose; (b) recommended with the step calendar published |
| D5 | before the next site push | The chip claim: mixer x8 measured at 0.92x with the 3x factor, "under 2x" holds with margin on the stated convention | Confirm the claim stays, with the convention named |
| D6 | Counter ASIC 3.0 | AMD mines at a seventh of a 5090 and 4.9x the electricity per hash; no read width closes it | Accept for v3; the miners page says so |
| D11 | before the next site push | The hero, the abstract and the level 1 copy now say "the model and the bounty are public" / "bounty standing"; funding.md prices the chip bounty at USD 50,000, marks it NOT FUNDED, and its rule 3 says a bounty is announced only when escrowed | Strike "and the bounty" from the hero and the abstract until the USD 50,000 is escrowed, or escrow it; the litepaper's older "a bounty is attached" (finality, line 686) is the same question |
| D10 | after route A's rows and D9's card | Proving route: route A (re-sized SP1 server) now; RISC Zero as a second proof family (Apple, and the fallback) is a consensus and verifier change | Measure both; adopt a second family only on your say |
| # | Row | What needs deciding | Recommendation | Why |
|---|---|---|---|---|
| D1 | C1 | The fee switch lands at H = 210,000 about 19:50Z on 6 October, and the 0.3.10 node's export does not carry the switch, so every app prover's shards are refused or vetoed from H. Move H, or race 0.3.11 onto every prover first | If 0.3.11 (with the proving-v1 fork, whose export carries `daaScore` and `feesV1ActivationDaa`) is not on every prover by 16:00Z on 6 October, republish the override with H = tip + 86,400 rounded to the next thousand, re-read the digest on a scratch node, every node in one sweep (the fee-switch plan's own rule for a later H). The coordinator holds the devnet delegation; this note is so the morning does not find proving dark | A dark proving pool on the devnet costs nothing on chain (the escrow keeps it) but every coverage, latency and fleet number measured after H is void |
| D2 | C3 | The litepaper says "12 GB or more proves full shards"; the only measurement puts the prover alone at 13.8 GB on a 32 GB card | The sweep and the S_p curve are in (proving agent, 5 October late): no knob moves the floor; 12 GB proves nothing on SP1 6.8.1, 16 GB proves only empty shards alone, 24 GB proves the adopted 30,000-pgas shard (20.4 GB alone, about 22 GB beside the miner), 32 GB proves everything. Change the sentence to "24 GB or more proves; 32 GB proves and mines on one card" and evidence row 16 to tested-by-the-team on those rows; the 12 GB figure returns only if a smaller GPU server or a smaller shard measures under 12 GB. The proving agent has drafted the two replacement sentences on its branch (app 440fd59, litepaper and evidence rows 15 and 16); nothing is pushed, so the project lead approves or rewrites the draft rather than starting from the measurement. Two drafts exist: the proving agent's litepaper sentences (proving-v1 440fd59) and the coordinator's site, litepaper and miner-page line (ca2-coord 1c8439f); the integrator keeps one. One marker is owed on either: the 24 GB figure is the 5090's allocation pattern on a 32 GB card (22.2 GB beside the miner), not a measurement on a 24 GB card, and evidence rule 3 wants that said until a 4090 or a 5080-class 24 GB card runs it | A public number that the first 3060 owner disproves is the FUD the ledger exists to prevent |
| D3 | C8, C13 | The homepage says "Any 4 GB card" and the litepaper says a 4 GB card mines for about four years and an 8 GB card for more than a decade; the tile says "4B hard cap" while the rule mints about 3.96 billion | Replace with the lifetime table's numbers once the sub-agent lands it: "4 GB cards mine at launch; 8 GB for about six years; 12 GB for about fourteen; the dataset grows half a gigabyte a year" and "cap 4 billion, about 3.96 billion ever minted" | The schedule is public and the arithmetic is one line; a reader will do it |
| D4 | C9, C8 | Index mapping at gate 1, now with the lifetime table (`docs/analysis/card-lifetime-2026-10-05.md`): (a) multiply-shift, continuous growth: 4 GB cards out at 1.0 to 1.5 years, 8 GB at 6.3 to 7.5, 12 GB at 12 to 13.5, Apple 8 GB at 2.6 to 3.4; (b) power-of-two steps at years 4, 12, 28, 60: 4 GB to year 4, 8 GB to year 12, 12 GB to year 28 (the cache freed after the build, decided by the coordinator at 22:50), each tier ending on a step day | (b), with the step calendar published on the miners page from day one (the years 4, 12, 28 and 60 named beside the tiers), and the 1 GiB vectors kept. Revised from (a) at 23:0x: the table shows (b) is the only mapping under which the litepaper's "4 GB about four years, 8 GB more than a decade" holds, and a step day known twelve years ahead is a schedule, not an event; the coordinator recommends the same | Under (a) both public sentences are wrong today by 2.5 to 5 years; under (b) they hold and the cliffs are dated |
| D5 | C10 | The site's "under 2x" chip claim: the scratch layer leaves the on-die-cache recompute chip at 2.4x at every share; the mixer multiplier x2 brings it to 1.2x at twice the CPU verify cost (about 0.9 ms per warp, half the pool members per core) | Overtaken by the coordinator's delegated decisions (21:50 to 21:28 Mac clock): the public claim was qualified, the mixer x4 is in class v3 (chip row 0.61x bare, 1.84x with a 3x fixed-function factor, 1.53x at equal silicon with the mirror deducted: "under 2x" holds on the equal-silicon convention and is thin), and x8 is built and measured beside it (0.92x with the factor, from the M16 table); x8 goes in if the verify stays under 10 ms and the daily build under 1 s on every card we own (C23 asks that the iGPU tier be named in that rule). Measured at 21:40 UTC (ca2-mixer 54bbfcc): x8 is 2.1x the v2 verifier (2.79 ms per warp on a loaded core, 1.26 quiet by scaling), 7.1 ms of the 10 ms gate left, the Mac 1 GiB build flat at 21 ms; the verify half of the x8 rule passes, the PC build rows are owed. Then at 22:06 UTC x8 was decided into v3 on the PC build rows (latency-bound on every card): the chip row reads 0.92x with the 3x fixed-function factor, so "under 2x" holds with margin and the level 3 page can state the convention and the margin. For the project lead: confirm "under 2x" stays, now with the measured row behind it | The claim is the project's first public sentence on chips; it is either true by a measured lever or it is marked |
| D6 | C11 | Whether class v3 should favour AMD (fewer, wider loads per hash) at a cost to the 5090's latency-bound share, or accept that AMD cards mine at about a seventh of a 5090 and 4.5x the electricity per hash | Accept it for v3 and say so on the miners page ("NVIDIA first; AMD mines at a lower rate per watt on this class"); open the AMD question as a Counter ASIC 3.0 item with its own measurement | Tonight's rule was the project lead's and the 5090 margin is the anti-chip argument; AMD's position is a public-copy question, not a gate |
| D7 | C13 | Same as D3's supply wording | with D3 | |
| D8 | C7 | The miner page says "One click: install, start, the card mines and proves" (site/miner.html 7, 13, 21; site/index.html 484). On HiveOS and the rig installer a rig mines only until a Linux prover build is published, and on the app a 12 GB card mines only, a 16 GB card proves with the miner paused, 20 GB and up does both (the sweep of 5 October, before the v1-shard row) | Qualify the sentence on the miner page and the homepage card: "the card mines; 24 GB cards prove too, 32 GB does both at once; rigs mine until the Linux prover ships". The same page's "a visible 1% software fee you can switch off" becomes "solo mining carries a 1% software fee you can switch off; in a pool the pool's own fee is the only one" once pool-v0 ships (the pool agent fixed the rule: no software dev fee in pool mode) | The sentence is the product's first promise and tonight's measurement bounds it by card |
| D9 | C3, C15, C26 | No 12 GB NVIDIA card exists in the fleet, so every 12 GB number tonight is scaled from the 5090 and labelled approximate. The 13.9 GB floor is SP1's GPU server code (`docs/analysis/proving-methods.md`, branch proving-methods e7e0db7, section 1.3: builder.rs adds 4 GB to the card's physical memory and panics under 24, so a 16 GB card (16 + 4 = 20) and a 12 GB card never start and 20 GB is the smallest that does; the trace is allocated at the maximum shard; the CUDA mempool never releases; the proving plan's "under 24" (4c82e56) is the same test read before the addition). The prover-floor patch and the per-card profiles cannot be measured without the hardware | Buy one 12 GB NVIDIA card this week for PC 1's spare slot (an RTX 3060 12 GB or 4070 12 GB, about £250 to £400, approximate; the coordinator's request). Check first whether the RTX 3060 and the RTX 5060 Ti 16 GB that evidence.md row 16 says are "on order" are real and arriving; if so, no purchase, only the delivery date. No public line says "12 GB proves" before a real 12 GB card runs the rebuilt server on the S_p-curve fixtures and recipe | Money, and the one measurement every 12 GB claim rests on; the prover-floor agent's rows tonight replace the approximate figures when they land |
| D10 | C3, C15, C26, C28 | The proving route for the 12 GB tier and for Apple: `docs/analysis/proving-methods.md` section 4 ranks (1) route A, a re-sized SP1 GPU server with `S_p` as the dial and one server per card on rigs, nothing in consensus moving; (2) route D, RISC Zero 3.0.x as proof-system version 2 (the only shipped prover with a documented sub-12 GB configuration and a Metal path), the fallback if A misses 11 GB and the Apple route either way, 3 to 4 days plus a 3-month two-verifier overlap and three spec 7.8 rules; (3) a sumcheck family without a codeword (Jolt-class) in years, not now. Route A is already running tonight; route D adds a second proof family to the node, which is a consensus and verifier change outside tonight's delegation | Route A on the prover-floor rows, gated as the document says (the adopted shard under 11 GB alone and under 60 s prove-only on a real 12 GB card, D9); route D's measurement (RISC Zero at po2 19 and 20 on PC 2 and the Mac's Metal row) may run as a measurement, but adopting a second proof family waits for the project lead and for route A's result; the public paragraph of section 5 ("Proving runs on NVIDIA cards with 24 GB or more today. A build for 12 GB and 16 GB cards is being measured ...") is the honest line meanwhile and is the one of the three drafts to prefer, because it names what is being measured instead of a tier | A second verifier in the node is the kind of change the testnet's genesis must carry from day one; measuring it costs nothing, adopting it is the project lead's |
| D11 | C29 context; `docs/plans/funding.md` 36 and 63 | The chip bounty on the public pages: "the model and the bounty are public" (hero, abstract, ca2-coord cabec3b), "bounty standing" (level 1 copy); funding.md: USD 50,000 standing, "Not funded", "a bounty is announced only when it is escrowed"; the litepaper already says "a bounty is attached" to the finality review (line 686) | Strike "and the bounty" from the hero and the abstract and "standing" from level 1 until the money is escrowed, and say "a bounty follows the external review" if a sentence is wanted; or escrow USD 50,000 (and the USD 25,000 finality bounty) and keep the words. Applied at 22:25 (ca2-coord 7e6f77c): "and the bounty" struck from the hero and the abstract, "standing" from level 1, the copy says "a bounty follows the external review"; the words return only once escrowed | A public promise of money the project has not set aside is the FUD the ledger exists to prevent, by the project's own funding rule |

View file

@ -0,0 +1,381 @@
{
"pass": true,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1f010000",
"template_switch": {
"epoch": 3,
"daa": 180,
"at": 168
},
"run_ended_at_s": 298.3,
"final_daa": 300,
"blocks": {
"total": 305,
"before_boundary": 181,
"after_boundary": 124,
"chain_before": 176,
"chain_after": 123
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "8f8806638d59850f",
"seed": "234e082d653dc69d",
"miners": 3,
"disagree": false,
"ready_ms": [
219,
235,
232
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "fd9562df32a68313",
"seed": "9b36731951ad7fb3",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "1ae6d90ab299154c",
"seed": "ce88239dfe686691",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "5d0dedd9fd9e29a1",
"seed": "f177ab855facbeb4",
"miners": 3,
"disagree": false,
"ready_ms": [
185,
179,
183,
177,
183,
177
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "e81808dcdb02ce05",
"seed": "bcbc22513d3b73a2",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "06aff9c1d33e7a13",
"seed": "8d6755bc0eb15465",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
}
],
"accepted_per_miner": [
97,
103,
104
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"082fd39ba65df2ff",
"082fd39ba65df2ff",
"082fd39ba65df2ff"
],
"block_counts": [
304,
304,
304
],
"tips_per_node": [
1,
1,
1
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.7,
"daa": 0,
"epoch": 0,
"class": 2,
"nodes": [
"0/234e082d",
"0/234e082d",
"0/234e082d"
]
},
{
"t": 19.7,
"daa": 11,
"epoch": 0,
"class": 2,
"nodes": [
"11/c036063c",
"11/c036063c",
"11/c036063c"
]
},
{
"t": 34.7,
"daa": 34,
"epoch": 0,
"class": 2,
"nodes": [
"34/62ae0e8e",
"34/62ae0e8e",
"34/62ae0e8e"
]
},
{
"t": 49.8,
"daa": 53,
"epoch": 0,
"class": 2,
"nodes": [
"53/c21d7e86",
"53/c21d7e86",
"53/c21d7e86"
]
},
{
"t": 64.8,
"daa": 70,
"epoch": 1,
"class": 2,
"nodes": [
"70/9cb940da",
"70/9cb940da",
"70/9cb940da"
]
},
{
"t": 79.8,
"daa": 88,
"epoch": 1,
"class": 2,
"nodes": [
"88/c84b8514",
"88/c84b8514",
"88/c84b8514"
]
},
{
"t": 94.9,
"daa": 103,
"epoch": 1,
"class": 2,
"nodes": [
"103/1e2dfed3",
"103/1e2dfed3",
"103/1e2dfed3"
]
},
{
"t": 109.9,
"daa": 111,
"epoch": 1,
"class": 2,
"nodes": [
"111/aa7ea4cd",
"111/aa7ea4cd",
"111/aa7ea4cd"
]
},
{
"t": 124.9,
"daa": 139,
"epoch": 2,
"class": 2,
"nodes": [
"139/d52be92d",
"139/d52be92d",
"139/d52be92d"
]
},
{
"t": 139.9,
"daa": 153,
"epoch": 2,
"class": 2,
"nodes": [
"153/02a4323f",
"153/02a4323f",
"153/02a4323f"
]
},
{
"t": 155,
"daa": 165,
"epoch": 2,
"class": 2,
"nodes": [
"165/f2e38b3d",
"165/f2e38b3d",
"165/f2e38b3d"
]
},
{
"t": 170,
"daa": 183,
"epoch": 3,
"class": 3,
"nodes": [
"183/98f2ca9d",
"183/98f2ca9d",
"183/98f2ca9d"
]
},
{
"t": 185.1,
"daa": 195,
"epoch": 3,
"class": 3,
"nodes": [
"195/eecbf07e",
"195/eecbf07e",
"195/eecbf07e"
]
},
{
"t": 200.1,
"daa": 205,
"epoch": 3,
"class": 3,
"nodes": [
"205/ad4a2646",
"205/ad4a2646",
"205/ad4a2646"
]
},
{
"t": 215.1,
"daa": 221,
"epoch": 3,
"class": 3,
"nodes": [
"221/f8626ffd",
"221/f8626ffd",
"221/f8626ffd"
]
},
{
"t": 230.2,
"daa": 233,
"epoch": 3,
"class": 3,
"nodes": [
"233/d94e1815",
"233/d94e1815",
"233/d94e1815"
]
},
{
"t": 245.2,
"daa": 251,
"epoch": 4,
"class": 3,
"nodes": [
"251/f724fd9e",
"251/f724fd9e",
"251/f724fd9e"
]
},
{
"t": 260.2,
"daa": 266,
"epoch": 4,
"class": 3,
"nodes": [
"266/b91936fd",
"266/b91936fd",
"266/b91936fd"
]
},
{
"t": 275.2,
"daa": 276,
"epoch": 4,
"class": 3,
"nodes": [
"276/7c8f720d",
"276/7c8f720d",
"276/7c8f720d"
]
},
{
"t": 290.3,
"daa": 297,
"epoch": 4,
"class": 3,
"nodes": [
"297/63eb2574",
"297/63eb2574",
"297/63eb2574"
]
}
]
}

View file

@ -0,0 +1,378 @@
{
"pass": true,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1f010000",
"template_switch": {
"epoch": 3,
"daa": 180,
"at": 171
},
"run_ended_at_s": 292.3,
"final_daa": 300,
"blocks": {
"total": 305,
"before_boundary": 181,
"after_boundary": 124,
"chain_before": 180,
"chain_after": 122
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "8f8806638d59850f",
"seed": "234e082d653dc69d",
"miners": 3,
"disagree": false,
"ready_ms": [
177,
177,
178
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "bb8dd9ddbf9eb63f",
"seed": "2f66af56da44ca92",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "8ee7a9f33d418e48",
"seed": "5194dfa8a53259bc",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "2d278041ba482dba",
"seed": "54c353de1d8609d7",
"miners": 3,
"disagree": false,
"ready_ms": [
181,
179,
179
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "2ae786d294a8a59d",
"seed": "8529c69223a2194c",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "bc36813df2f41b5f",
"seed": "fdc233b69c42ffea",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
}
],
"accepted_per_miner": [
102,
102,
100
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"712c1b212091dcdc",
"712c1b212091dcdc",
"712c1b212091dcdc"
],
"block_counts": [
303,
303,
303
],
"tips_per_node": [
1,
1,
1
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.6,
"daa": 0,
"epoch": 0,
"class": 2,
"nodes": [
"0/234e082d",
"0/234e082d",
"0/234e082d"
]
},
{
"t": 19.7,
"daa": 18,
"epoch": 0,
"class": 2,
"nodes": [
"18/5b5f2d36",
"18/5b5f2d36",
"18/5b5f2d36"
]
},
{
"t": 34.7,
"daa": 51,
"epoch": 0,
"class": 2,
"nodes": [
"51/333c61cc",
"51/333c61cc",
"51/333c61cc"
]
},
{
"t": 49.8,
"daa": 67,
"epoch": 1,
"class": 2,
"nodes": [
"67/f1fcbc43",
"67/f1fcbc43",
"67/f1fcbc43"
]
},
{
"t": 64.8,
"daa": 89,
"epoch": 1,
"class": 2,
"nodes": [
"89/91180594",
"89/91180594",
"89/91180594"
]
},
{
"t": 79.8,
"daa": 101,
"epoch": 1,
"class": 2,
"nodes": [
"101/49dd0422",
"101/49dd0422",
"101/49dd0422"
]
},
{
"t": 94.9,
"daa": 111,
"epoch": 1,
"class": 2,
"nodes": [
"111/a65a10b3",
"111/a65a10b3",
"111/a65a10b3"
]
},
{
"t": 109.9,
"daa": 122,
"epoch": 2,
"class": 2,
"nodes": [
"122/f06b1eb3",
"122/f06b1eb3",
"122/f06b1eb3"
]
},
{
"t": 124.9,
"daa": 140,
"epoch": 2,
"class": 2,
"nodes": [
"140/bbfe6710",
"140/bbfe6710",
"140/bbfe6710"
]
},
{
"t": 140,
"daa": 155,
"epoch": 2,
"class": 2,
"nodes": [
"155/a4e736dd",
"155/a4e736dd",
"155/a4e736dd"
]
},
{
"t": 155,
"daa": 171,
"epoch": 2,
"class": 2,
"nodes": [
"171/c754adeb",
"171/c754adeb",
"171/c754adeb"
]
},
{
"t": 170,
"daa": 179,
"epoch": 2,
"class": 2,
"nodes": [
"179/da89ab4d",
"179/da89ab4d",
"179/da89ab4d"
]
},
{
"t": 185,
"daa": 187,
"epoch": 3,
"class": 3,
"nodes": [
"187/e0f61041",
"187/e0f61041",
"187/e0f61041"
]
},
{
"t": 200.1,
"daa": 204,
"epoch": 3,
"class": 3,
"nodes": [
"204/5d37e885",
"204/5d37e885",
"204/5d37e885"
]
},
{
"t": 215.1,
"daa": 222,
"epoch": 3,
"class": 3,
"nodes": [
"222/46643165",
"222/46643165",
"222/46643165"
]
},
{
"t": 230.1,
"daa": 236,
"epoch": 3,
"class": 3,
"nodes": [
"236/17eb5a95",
"236/17eb5a95",
"236/17eb5a95"
]
},
{
"t": 245.2,
"daa": 247,
"epoch": 4,
"class": 3,
"nodes": [
"247/7aada19d",
"247/7aada19d",
"247/7aada19d"
]
},
{
"t": 260.2,
"daa": 261,
"epoch": 4,
"class": 3,
"nodes": [
"261/15d8f83e",
"261/15d8f83e",
"261/15d8f83e"
]
},
{
"t": 275.3,
"daa": 274,
"epoch": 4,
"class": 3,
"nodes": [
"274/9f3d19c5",
"274/9f3d19c5",
"274/9f3d19c5"
]
},
{
"t": 290.3,
"daa": 297,
"epoch": 4,
"class": 3,
"nodes": [
"297/fa7975bd",
"297/fa7975bd",
"297/fa7975bd"
]
}
]
}

View file

@ -0,0 +1,367 @@
{
"pass": true,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1f010000",
"template_switch": {
"epoch": 3,
"daa": 181,
"at": 127.9
},
"run_ended_at_s": 280.2,
"final_daa": 300,
"blocks": {
"total": 304,
"before_boundary": 182,
"after_boundary": 122,
"chain_before": 180,
"chain_after": 120
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "8f8806638d59850f",
"seed": "234e082d653dc69d",
"miners": 3,
"disagree": false,
"ready_ms": [
179,
178,
179
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "e145305446a6b4ce",
"seed": "09952ab515cc5a10",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "743ad2a3cab0518a",
"seed": "1812a8eabb5a452c",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "a6523b90cff501e3",
"seed": "6296e3d38df15872",
"miners": 3,
"disagree": false,
"ready_ms": [
191,
191,
191
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "bc811b3c4b8b1ced",
"seed": "a59152f149a517e0",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "e784541f19cdebe5",
"seed": "87ec7849fce0019f",
"miners": 3,
"disagree": false,
"ready_ms": [
2,
2,
2
]
}
],
"accepted_per_miner": [
101,
89,
113
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"a9ce45df8beeaf13",
"a9ce45df8beeaf13",
"a9ce45df8beeaf13"
],
"block_counts": [
303,
303,
303
],
"tips_per_node": [
1,
1,
1
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.6,
"daa": 0,
"epoch": 0,
"class": 2,
"nodes": [
"0/234e082d",
"0/234e082d",
"0/234e082d"
]
},
{
"t": 19.7,
"daa": 48,
"epoch": 0,
"class": 2,
"nodes": [
"48/78df0834",
"48/78df0834",
"48/78df0834"
]
},
{
"t": 34.7,
"daa": 67,
"epoch": 1,
"class": 2,
"nodes": [
"67/0b934b49",
"67/0b934b49",
"67/0b934b49"
]
},
{
"t": 49.7,
"daa": 84,
"epoch": 1,
"class": 2,
"nodes": [
"84/2ba5b427",
"84/2ba5b427",
"84/2ba5b427"
]
},
{
"t": 64.8,
"daa": 101,
"epoch": 1,
"class": 2,
"nodes": [
"101/6574b577",
"101/6574b577",
"101/6574b577"
]
},
{
"t": 79.8,
"daa": 123,
"epoch": 2,
"class": 2,
"nodes": [
"123/21e2b6ce",
"123/21e2b6ce",
"123/21e2b6ce"
]
},
{
"t": 94.8,
"daa": 134,
"epoch": 2,
"class": 2,
"nodes": [
"134/74b83d55",
"134/74b83d55",
"134/74b83d55"
]
},
{
"t": 109.9,
"daa": 153,
"epoch": 2,
"class": 2,
"nodes": [
"153/71b251a6",
"153/71b251a6",
"153/71b251a6"
]
},
{
"t": 124.9,
"daa": 175,
"epoch": 2,
"class": 2,
"nodes": [
"175/b1500fb5",
"175/b1500fb5",
"175/b1500fb5"
]
},
{
"t": 139.9,
"daa": 185,
"epoch": 3,
"class": 3,
"nodes": [
"185/dd8ff66c",
"185/dd8ff66c",
"185/dd8ff66c"
]
},
{
"t": 155,
"daa": 191,
"epoch": 3,
"class": 3,
"nodes": [
"191/2ef77979",
"191/2ef77979",
"191/2ef77979"
]
},
{
"t": 170,
"daa": 194,
"epoch": 3,
"class": 3,
"nodes": [
"194/1871cd31",
"194/1871cd31",
"194/1871cd31"
]
},
{
"t": 185,
"daa": 206,
"epoch": 3,
"class": 3,
"nodes": [
"206/808070af",
"206/808070af",
"206/808070af"
]
},
{
"t": 200.1,
"daa": 221,
"epoch": 3,
"class": 3,
"nodes": [
"221/1bccc51c",
"221/1bccc51c",
"221/1bccc51c"
]
},
{
"t": 215.1,
"daa": 237,
"epoch": 3,
"class": 3,
"nodes": [
"237/d0bfaef8",
"237/d0bfaef8",
"237/d0bfaef8"
]
},
{
"t": 230.1,
"daa": 253,
"epoch": 4,
"class": 3,
"nodes": [
"253/a4136a32",
"253/a4136a32",
"253/a4136a32"
]
},
{
"t": 245.2,
"daa": 268,
"epoch": 4,
"class": 3,
"nodes": [
"268/51ef3166",
"268/51ef3166",
"268/51ef3166"
]
},
{
"t": 260.2,
"daa": 277,
"epoch": 4,
"class": 3,
"nodes": [
"277/869d0dd3",
"277/869d0dd3",
"277/869d0dd3"
]
},
{
"t": 275.2,
"daa": 292,
"epoch": 4,
"class": 3,
"nodes": [
"292/fe54acb8",
"292/fe54acb8",
"292/fe54acb8"
]
}
]
}

View file

@ -0,0 +1,474 @@
{
"pass": false,
"checks": {
"switch_line_on_every_node": true,
"switch_line_names_the_rounded_epoch": true,
"template_switched_at_the_first_v3_epoch": true,
"blocks_before_the_boundary": true,
"blocks_after_the_boundary": true,
"v2_and_v3_programs_seen": true,
"program_ids_differ_across_the_switch": true,
"miners_agree_on_every_program": true,
"zero_rejected_by_miners": true,
"zero_rejected_by_nodes": true,
"sinks_agree": true,
"block_counts_agree": true,
"metal_prepare_sent_for_v3": true,
"metal_worker_prepared_v3_pack": true,
"metal_no_need_or_mismatch": false,
"metal_accepted_blocks_after_switch": true,
"metal_cpu_recheck_clean": true,
"metal_swapped_without_pause": true
},
"activation": 150,
"epoch_blocks": 60,
"first_v3_epoch": 3,
"boundary_daa": 180,
"secs": 480,
"threads": 1,
"node": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneumd",
"miner": "/Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2/target-ca2/release/igneum-miner",
"genesis_bits": "0x1e010000",
"template_switch": {
"epoch": 3,
"daa": 180,
"at": 264.5
},
"run_ended_at_s": 366.8,
"final_daa": 301,
"blocks": {
"total": 305,
"before_boundary": 182,
"after_boundary": 123,
"chain_before": 91,
"chain_after": 67
},
"programs": [
{
"epoch": 0,
"class": "v2",
"program_id": "3e974094b5ce543f",
"seed": "130e3b7d568db07a",
"miners": 2,
"disagree": false,
"ready_ms": [
192,
192
]
},
{
"epoch": 1,
"class": "v2",
"program_id": "1d5b129e89ec2995",
"seed": "bc056261dafe0ec2",
"miners": 2,
"disagree": false,
"ready_ms": [
4,
4
]
},
{
"epoch": 2,
"class": "v2",
"program_id": "835e953cdc2c11be",
"seed": "1908705502833bdf",
"miners": 2,
"disagree": false,
"ready_ms": [
9,
7
]
},
{
"epoch": 3,
"class": "v3",
"program_id": "082b7ee882aaf418",
"seed": "4ae011065b84f679",
"miners": 2,
"disagree": false,
"ready_ms": [
1069,
1085
]
},
{
"epoch": 4,
"class": "v3",
"program_id": "7ef81e08e4611665",
"seed": "d5964a745dbe2ce2",
"miners": 2,
"disagree": false,
"ready_ms": [
6,
6
]
},
{
"epoch": 5,
"class": "v3",
"program_id": "ecee6312983c15ee",
"seed": "f9d5de9eea7330ae",
"miners": 2,
"disagree": false,
"ready_ms": [
5,
5
]
}
],
"accepted_per_miner": [
301,
2,
1
],
"rejected_by_miners": [
0,
0,
0
],
"rejected_by_nodes": [
0,
0,
0
],
"rejected_lines": [],
"sinks": [
"66a33eca15d33ba2",
"66a33eca15d33ba2",
"66a33eca15d33ba2"
],
"block_counts": [
304,
304,
304
],
"tips_per_node": [
2,
2,
2
],
"switch_lines": [
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)",
"Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)"
],
"samples": [
{
"t": 4.6,
"daa": 1,
"epoch": 0,
"class": 2,
"nodes": [
"1/97755438",
"0/130e3b7d",
"0/130e3b7d"
]
},
{
"t": 19.7,
"daa": 3,
"epoch": 0,
"class": 2,
"nodes": [
"3/52e2c440",
"3/52e2c440",
"3/52e2c440"
]
},
{
"t": 34.7,
"daa": 3,
"epoch": 0,
"class": 2,
"nodes": [
"3/52e2c440",
"3/52e2c440",
"3/52e2c440"
]
},
{
"t": 49.7,
"daa": 13,
"epoch": 0,
"class": 2,
"nodes": [
"13/fa702087",
"13/fa702087",
"13/fa702087"
]
},
{
"t": 64.8,
"daa": 38,
"epoch": 0,
"class": 2,
"nodes": [
"38/2a687376",
"38/2a687376",
"38/2a687376"
]
},
{
"t": 79.8,
"daa": 55,
"epoch": 0,
"class": 2,
"nodes": [
"55/2f1f0ea8",
"55/2f1f0ea8",
"55/2f1f0ea8"
]
},
{
"t": 94.8,
"daa": 57,
"epoch": 0,
"class": 2,
"nodes": [
"57/e1b2abc1",
"57/e1b2abc1",
"57/e1b2abc1"
]
},
{
"t": 109.9,
"daa": 61,
"epoch": 1,
"class": 2,
"nodes": [
"61/86fa7300",
"61/86fa7300",
"61/86fa7300"
]
},
{
"t": 124.9,
"daa": 61,
"epoch": 1,
"class": 2,
"nodes": [
"61/86fa7300",
"61/86fa7300",
"61/86fa7300"
]
},
{
"t": 140,
"daa": 61,
"epoch": 1,
"class": 2,
"nodes": [
"61/86fa7300",
"61/86fa7300",
"61/86fa7300"
]
},
{
"t": 155,
"daa": 73,
"epoch": 1,
"class": 2,
"nodes": [
"73/e7f6e171",
"73/e7f6e171",
"73/e7f6e171"
]
},
{
"t": 170.1,
"daa": 94,
"epoch": 1,
"class": 2,
"nodes": [
"94/0b1c93b1",
"94/0b1c93b1",
"94/0b1c93b1"
]
},
{
"t": 185.2,
"daa": 114,
"epoch": 1,
"class": 2,
"nodes": [
"114/16034702",
"114/16034702",
"114/16034702"
]
},
{
"t": 200.2,
"daa": 117,
"epoch": 1,
"class": 2,
"nodes": [
"117/d018a345",
"117/d018a345",
"117/d018a345"
]
},
{
"t": 215.4,
"daa": 120,
"epoch": 2,
"class": 2,
"nodes": [
"120/e022d64e",
"120/e022d64e",
"120/e022d64e"
]
},
{
"t": 230.4,
"daa": 134,
"epoch": 2,
"class": 2,
"nodes": [
"134/e7d07203",
"134/e7d07203",
"134/e7d07203"
]
},
{
"t": 245.4,
"daa": 157,
"epoch": 2,
"class": 2,
"nodes": [
"157/d23b3a1b",
"157/d23b3a1b",
"157/d23b3a1b"
]
},
{
"t": 260.5,
"daa": 175,
"epoch": 2,
"class": 2,
"nodes": [
"175/4bf5acbe",
"175/4bf5acbe",
"175/4bf5acbe"
]
},
{
"t": 275.5,
"daa": 198,
"epoch": 3,
"class": 3,
"nodes": [
"198/f337dbc1",
"198/f337dbc1",
"198/f337dbc1"
]
},
{
"t": 290.6,
"daa": 217,
"epoch": 3,
"class": 3,
"nodes": [
"217/dc28325d",
"217/dc28325d",
"217/dc28325d"
]
},
{
"t": 305.6,
"daa": 234,
"epoch": 3,
"class": 3,
"nodes": [
"234/a0079166",
"234/a0079166",
"234/a0079166"
]
},
{
"t": 320.7,
"daa": 249,
"epoch": 4,
"class": 3,
"nodes": [
"249/eb206cb2",
"249/eb206cb2",
"249/eb206cb2"
]
},
{
"t": 335.7,
"daa": 268,
"epoch": 4,
"class": 3,
"nodes": [
"268/9b9866b1",
"268/9b9866b1",
"268/9b9866b1"
]
},
{
"t": 350.7,
"daa": 285,
"epoch": 4,
"class": 3,
"nodes": [
"285/3263ff7a",
"285/3263ff7a",
"285/3263ff7a"
]
},
{
"t": 365.8,
"daa": 299,
"epoch": 4,
"class": 3,
"nodes": [
"299/31eff7d0",
"299/31eff7d0",
"299/31eff7d0"
]
}
],
"metal": {
"worker": "/Users/joshm/Projects/igneum-wt-ca2-v3/proto-metal/igneum-bench-ca2",
"prepares": [
"PREPARE sent for epoch seed b7eda892cff02b39c043f5fa683a7c1629c96d02fa480522603be595372e371f day 1243916 class v2 (6 DAA blocks before the boundary at 60; CPU side ready in 960 ms)",
"PREPARE sent for epoch seed bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c day 1243916 class v2 (the current pair, after seed mismatches; CPU side ready in 1248 ms)",
"PREPARE sent for epoch seed 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3 day 1243916 class v2 (6 DAA blocks before the boundary at 120; CPU side ready in 1768 ms)",
"PREPARE sent for epoch seed 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c day 1243916 class v3 (6 DAA blocks before the boundary at 180; CPU side ready in 1740 ms)",
"PREPARE sent for epoch seed d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 day 1243916 class v3 (6 DAA blocks before the boundary at 240; CPU side ready in 1289 ms)",
"PREPARE sent for epoch seed f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e day 1243916 class v3 (7 DAA blocks before the boundary at 300; CPU side ready in 1166 ms)"
],
"prepared": [
"worker: prepared b7eda892cff02b39c043f5fa683a7c1629c96d02fa480522603be595372e371f 69676e65756d2d6461792f0cfb120000000000 35663.4 program 64.5 dataset 0.0 race 35598.9 variant base class v2 loads/hash 128 cache-fill 2.0 build 30.5 resident 2 programs 1 datasets (prepare answered 1.3 s after it was sent)",
"worker: prepared bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c 69676e65756d2d6461792f0cfb120000000000 33632.6 program 108.1 dataset 0.0 race 33524.5 variant base class v2 loads/hash 128 cache-fill 2.0 build 30.5 resident 2 programs 1 datasets (prepare answered 35.2 s after it was sent)",
"worker: prepared 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3 69676e65756d2d6461792f0cfb120000000000 34728.4 program 77.3 dataset 0.0 race 34651.0 variant base class v2 loads/hash 128 cache-fill 2.0 build 30.5 resident 2 programs 1 datasets (prepare answered 0.0 s after it was sent)",
"worker: prepared 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c 69676e65756d2d6461792f0cfb120000000000 252.5 program 55.5 dataset 196.9 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.3 s after it was sent)",
"worker: prepared d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 69676e65756d2d6461792f0cfb120000000000 51.1 program 51.1 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.1 s after it was sent)",
"worker: prepared f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e 69676e65756d2d6461792f0cfb120000000000 46.0 program 46.0 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.0 s after it was sent)"
],
"prepared_v3": [
"worker: prepared 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c 69676e65756d2d6461792f0cfb120000000000 252.5 program 55.5 dataset 196.9 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.3 s after it was sent)",
"worker: prepared d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 69676e65756d2d6461792f0cfb120000000000 51.1 program 51.1 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.1 s after it was sent)",
"worker: prepared f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e 69676e65756d2d6461792f0cfb120000000000 46.0 program 46.0 dataset 0.0 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1 resident 2 programs 2 datasets (prepare answered 0.0 s after it was sent)"
],
"need": 0,
"mismatch_lines": [
"1791239226.967 PACK OUT OF DATE: the prepared pair b7eda892cff02b39 is not the node's epoch bc056261dafe0ec2; rebuilding the program pack for the worker"
],
"refused": [],
"swaps": [
"SEED CHANGE at daa 60: epoch seed 130e3b7d568db07aad638f67b2ac26e7cf93c51e4f564d45c6724fc4b2d3da65 -> bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c, day 1243916 -> 1243916 (epoch 1): the pair was not prepared (unexpected seeds); the worker compiles inline",
"SEED CHANGE at daa 120: epoch seed bc056261dafe0ec266f2e8f508e8fe7349e4d0c064853c6fd37ffa6a5e1fab3c -> 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3, day 1243916 -> 1243916 (epoch 2): prepare was sent but the worker has not answered prepared yet; it compiles inline",
"SEED CHANGE at daa 180: epoch seed 1908705502833bdf6c5d08b6d6ae3a364e3fe2bf0956504b20bc5071eecbdfc3 -> 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c, day 1243916 -> 1243916 (epoch 3): swapped with no pause (prepared 4 s ago, prepare took 252 ms)",
"SEED CHANGE at daa 240: epoch seed 4ae011065b84f67924d0c42ff63669e6124b6bddc4a7f71a3f1075b5891efd8c -> d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6, day 1243916 -> 1243916 (epoch 4): swapped with no pause (prepared 4 s ago, prepare took 51 ms)",
"SEED CHANGE at daa 300: epoch seed d5964a745dbe2ce2a88d6a4198a244996f1f4aba6cc1bc20841bb532e469fdd6 -> f9d5de9eea7330aeec83c0994b8929e61a2ac2f13dbae6462cceace3b1d6f24e, day 1243916 -> 1243916 (epoch 5): swapped with no pause (prepared 6 s ago, prepare took 46 ms)"
],
"accepted_total": 301,
"accepted_after_switch": 124,
"found_lines": 0,
"cpu_recheck_mismatched": 0,
"last_status": ""
}
}

View file

@ -0,0 +1,193 @@
# Counter ASIC 2.0: the node side (program class v3 as a height switch)
5 October 2026, night, worker "ca2-node". Branches: `ca2-v3` (main repository: the igneum-pow seam, the workers, the fast-time gate) and `ca2-v3-node` (the fork, from the 0.3.10 tip 21d4c73c plus pack-loop 05ef0fa3). Plan: `docs/plans/counter-asic-2-rollout.md`; status: `docs/plans/counter-asic-2-status.md`. The shape follows finality v3 (`docs/plans/finality-v3-rollout-devnet.md` section 6): one height switch read from the override file, every node carries the same object before the height.
Everything below is the SEAM. The class itself (`igneum_pow::V3_CLASS`) is a placeholder, w16 (`LoadClass::fixed(4, 16)`), that the integration branch replaces with the decided width, mix and scratch share; the ca2-era draw and the ca2-mixer item construction fill what `Epoch::from_chain_seeds` calls. Nothing here changes a v2 program, a pinned pack or any live node.
## 1. What changed, where
| Piece | What |
|---|---|
| igneum-pow `generator.rs` | `ProgramClass { V2, V3 }`, `V3_CLASS` (placeholder), `GENERATOR_VERSION_V3 = 3`, `generate_from_seed_bytes_program_class(label, seed, class, era)`; `Program::era_bytes`; a v3 program's id is `program_id(3, seed, attempt)` (spec 01 section 1.4.6) |
| igneum-pow `verify.rs` | `Epoch::from_chain_seeds(epoch, day, era, class, label)`, `Epoch::chain_program` (no cache fill), `Epoch::chain_dataset(day, class)` (the one entry the node's day cache goes through; today both classes build the same cache) |
| igneum-pow `emit.rs` | program.h: `IGNEUM_PROGRAM_CLASS "v3"` and `IGNEUM_ERA_SEED_HEX` beside `IGNEUM_GENERATOR 3`; program.json: `program_class`, `era_seed_bytes`; nothing on a v2 pack (`tests/packs.rs` diffs the pinned packs `igneum-genesis-mh` and `igneum-devnet-v4-epoch0`: identical) |
| igneum-pow `packcheck.rs` | `verify_pack_texts_chain` / `verify_pack_dir_chain(dir, epoch, day, want_class, want_era)`: `PackFault::WrongClass` for the wrong class, the wrong era, or a generator other than 2 or 3; `PackIdentity` carries `generator`, `class`, `era_hex` |
| `proto-cuda/nvrtc/packfile.h` | `pf_load` refuses a generator other than 2 or 3 (spec 1.4.5), reads the class (must match the generator) and the era; `pf_pack_class_ok`, `pf_class_token` (the one rule for the `class=` / `era=` tokens) |
| `proto-cuda/nvrtc/worker.cpp`, `proto-opencl/host.c` | a pair's identity includes its class and era when the line names them; right seeds and the wrong class answer `need <e> <d>` and `error <id> pack <dir>: program class mismatch ...`, so the miner prepares the pair from a pack of the right class; a prepare on a wrong-class pack fails in plain words |
| `proto-metal/main.swift` | refuses every `class=v3` line (the Swift generator is version 2; the integration adds 3) |
| fork `consensus/core/src/igneum.rs` | `ProgramClass`, `program_class_for_epoch_at(e, N4, L)`, `program_class_v3_first_epoch_at`, the process-wide activation (`install_program_class_v3_activation`, `program_class_for_epoch`, `program_class_at`), `POW_ERA_BLOCKS`, `POW_ERA_LEAD`, `pow_era_index`, `pow_era_seed_score`; `PowEpochInfo` + `program_class`, `next_program_class`, `program_class_v3_activation_daa`, `era_index`, `era_seed` (serde defaults: v2, never, none) |
| fork `consensus/core/src/config/params.rs` | `program_class_v3_activation_daa` in `Params` and `OverrideParams` (default never on devnet, simnet, mainnet; 0 on the testnet like every other switch), `override_params`, `From<Params>`, the digest (unconditionally, right after `finality_v3_activation_daa`), `Params::install_program_class_v3_activation`, `Params::program_class_v3_first_epoch`; tests `override_params_carry_the_program_class_v3_activation`, the digest test's 11th edit, `fast_time_60x_file_is_the_devnet_at_60x` (every field present) |
| fork `kaspad/src/daemon.rs` | `Program class v3 from the override file: active from epoch E (DAA score N4 rounded up to the epoch boundary at 3600*E, epochs of 3600 DAA)`; the activation installed next to the PoW schedule |
| fork `consensus/pow/src/igneum.rs` | `EpochSeeds { epoch, day, class, era }` (+ `EpochSeeds::v2`), day caches keyed on `(day, class)`, the program through `Epoch::chain_program`, the cache through `Epoch::chain_dataset`, `standalone_epoch` through `Epoch::from_chain_seeds`; test `program_class_v3_seeds_hash_their_own_program_over_their_own_cache` |
| fork `consensus/src/pipeline/header_processor/{processor,pre_ghostdag_validation}.rs` | the class of the header's epoch (`program_class_at(daa)`), `HeaderProcessor::era_seed` (the stand-in, memoised per era), `RuleError::EraSeedUnavailable` |
| fork `consensus/src/processes/pruning_proof/igneum_pow.rs` | `seeds_for`: the class of the epoch; era 0 = genesis; a later era is `PruningImportError::MissingEraSeed` (no era witness in the proof format yet) |
| fork `consensus/src/consensus/mod.rs` | `get_pow_epoch_info`: the class of this and the next epoch, the activation, the era index and seed (the era walk memoised once per era per process) |
| fork `rpc/core/src/model/message.rs`, `rpc/grpc/core/proto/rpc.proto` (fields 12 to 16), `rpc/grpc/core/src/convert/message.rs` | `RpcPowEpochInfo` + `program_class` (generator number), `next_program_class`, `program_class_v3_activation_daa`, `era_index`, `era_seed`; an old node's absent fields read as v2, never, none |
| fork `igneum/miner/src/main.rs` | seeds from the template (`seeds_from_info`), the activation installed from the template, the legacy seed walk keys the class on it; `next_pair` takes the next epoch's class; the job and prepare lines end with `class=v3 era=<hex>` for a v3 epoch; `seeds.txt` carries `program_class` and `era_seed_hex`; `write_pack_checked` checks class and era (`verify_pack_dir_chain`); the "program and 256 MiB cache ready" lines and `export-pack` print the class and the program id |
| `infra/fast-time/override-60x.json`, `tools/finality-attacks/redteam/override-60x-v3.json`, `infra/fast-time/README.md` | the field at never, the README row |
| `infra/fast-time/class-v3.mjs` | the G4 gate (section 5) |
## 2. The epoch-boundary rule
One epoch has one program (spec 01 section 1.12), so the switch keys on the EPOCH: epoch `e` is class v3 when `L * e >= N4` with `L` the live epoch length (`pow_epoch_blocks()`, 3,600 on the devnet, 60 on the fast-time profile). The first v3 epoch is `ceil(N4 / L)`; a height inside an epoch rounds UP to the next boundary and never splits an epoch between two programs. A block's class is a function of its DAA score alone (`program_class_at(daa)`), as its epoch seed is.
| N4 | L | first v3 epoch | first v3 DAA | the epoch before |
|---|---|---|---|---|
| 150 | 60 | 3 | 180 | epoch 2 (DAA 120 to 179) is v2 to its last block |
| 180 | 60 | 3 | 180 | |
| 181 | 60 | 4 | 240 | |
| 136,000 | 3,600 | 38 | 136,800 | epoch 37 (133,200 to 136,799) is v2 |
| 136,800 | 3,600 | 38 | 136,800 | |
| 0 | any | 0 | 0 | (the testnet) |
| never | any | none | | |
The unit tests `program_class_switch_rounds_up_to_the_epoch_boundary` (consensus-core) and `override_params_carry_the_program_class_v3_activation` (params) pin these rows and sweep every activation 0..399 at L = 60.
The digest: the field (and `pow_genesis_dataset_log2`) enters `consensus_digest` unconditionally, so the digest flips the moment a binary carrying the field (at never) runs, exactly as the finality v3 field did. This is intended: every node must carry the object before any node reaches the height, and a node without the field is refused at the handshake (rollout section 1, order step 1). Measured: the devnet digest with no override file moves from `9409dedac4bf9f0f...` (0.3.10) to `c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c` (0.3.11; the pinned value of `consensus_digest_keeps_the_0_3_5_value_until_the_fee_switch_is_set`), which is the expected digest of a scratch node at the publish.
## 3. The era stand-in
`E_n` of spec 04 section 4.4, until the 1-hour VDF is in the node (`docs/plans/era-layout.md` section 2):
| Era | `E_n` |
|---|---|
| 0 | the genesis block hash |
| n >= 1 | the hash of the last selected-chain block whose DAA score is below `15,552,000 n - 7,200` |
One function per path: `HeaderProcessor::era_seed` (the header's era from its DAA score, the walk down the selected parents from the header's selected parent, memoised per era in `era_seed_memo`), `Consensus::get_pow_epoch_info` (the same walk from the sink for the template, memoised once per era per process), `ProofSeeds::seeds_for` (era 0 only). The seed block is at least one era lead (7,200 DAA) below any header that uses it, past the merge depth (3,600), so one walk per era per process is sound; the walk itself is up to an era long on the first header of era `n >= 1` (about 15.5 million selected parents), which is why it is memoised and why the VDF should land before era 1 (180 days after genesis). A v2 program never reads the era; the placeholder v3 class does not either (the ca2-era draw will); the era is carried and recorded in the pack so a worker of the wrong era is refused from the first v3 build.
## 4. The job line, the prepare line, the pack
| Surface | Class v2 (today) | Class v3 |
|---|---|---|
| job line | `job <id> <prehash> <target> <start> <count> <epoch> <day>` | the same with ` class=v3 era=<64 hex>` at the end |
| prepare line | `prepare <epoch> <day> [<dir>]` | the same with ` class=v3 era=<64 hex>` at the end |
| `program.h` | `IGNEUM_GENERATOR 2` | `IGNEUM_GENERATOR 3`, `IGNEUM_PROGRAM_CLASS "v3"`, `IGNEUM_ERA_SEED_HEX "<64 hex>"` |
| `program.json` | `"generator": 2` | `"generator": 3`, `"program_class": "v3"`, `"era_seed_bytes": "<hex>"` |
| `seeds.txt` | `epoch_seed_hex`, `day_seed_hex`, `day_index` | plus `program_class v3`, `era_seed_hex <hex>` |
| template `pow_epoch` | `programClass 2` | `programClass 3`, `nextProgramClass`, `programClassV3ActivationDaa`, `eraIndex`, `eraSeed` |
A v3 program's identity is the pair (program id, era seed): the id covers the generator, the seed words and the attempt (spec 01 section 1.4.6, unchanged), so every era of one epoch seed shares one id, and the era seed, carried by the pack (`IGNEUM_ERA_SEED_HEX`) and the job line (`era=`), tells them apart. The workers and `packcheck` compare both.
A v2 line and a v2 pack are byte for byte what the workers read before this branch (the tokens are sent only for a v3 epoch), so a 0.3.10 worker on a 0.3.11 miner mines v2 epochs unchanged and refuses nothing until the switch; by the switch every worker is 0.3.11 (rollout order).
Refusals: `pf_load` refuses a generator that is not 2 or 3 (`error 0 pack <dir>: program pack generator N is not a generator version this worker runs (2 or 3)`, the exit-44 path of 05ef0fa3: the miner rebuilds the pack before the restart). A job of class v3 against a resident v2 pack of the same seeds answers `need <e> <d>` and `error <id> pack <dir>: program class mismatch: this pack is class v2, the job names class v3 (export the pack again)`; the miner's `need` handling prepares the pair again, `write_pack_checked` writes a v3 pack (checked with `verify_pack_dir_chain` before the worker hears of it), and the prepared v3 pair wins over the resident v2 pair because the pair identity now carries the class. The Metal worker answers `error <id> program class v3 is not implemented by this worker` until the Swift generator carries version 3.
## 5. The fast-time gate (rollout G4)
`node infra/fast-time/class-v3.mjs [--secs 420] [--activation 150] [--epochs-after 2]` under `tools/lock/with-lock.sh run`: three nodes on `override-60x.json` merged with `genesis_bits` 0x1f010000 (2^16 hashes per block, the CPU difficulty of `sim/difficulty/testnet_v2.py`) and `program_class_v3_activation_daa` 150 (inside epoch 2, so the rounding rule is exercised: the first v3 epoch is 3 at DAA 180); one real CPU miner per node (`--engine igneum-pow`, 1 thread, real lottery-hash solutions, every node verifying the other two); the run ends two epochs after the boundary.
### Result, run 1 (5 October 2026, 21:33:04Z to 21:38:02Z, Apple M5 Max shared with other agents' builds)
Binaries: fork ca2-v3-node 79bd8e10 and igneum-pow at ca2-v3 66eeba3 (the mixer-x4 class, `V3_CLASS` = MX4, before the era and hot-table fields), `target-ca2/release`, built on the Mac under the build lock. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2133Z-mx4.json`. PASS: every check true.
| Check | Measured |
|---|---|
| Switch line on every node | 3 of 3: `Program class v3 from the override file: active from epoch 3 (DAA score 150 rounded up to the epoch boundary at 180, epochs of 60 DAA)` |
| Digest, all three nodes | `0186df7d0834d054...` (the 60x profile with the CPU bits and the switch) |
| Template class per epoch | epochs 0 to 2 class 2 (`nextProgramClass` 3 from epoch 2), epochs 3 to 5 class 3; the switch seen at DAA 180, 168.0 s wall |
| Blocks before / after the boundary (node 0's DAG) | 181 / 124 (selected chain 176 / 123), 305 in all |
| Program id per epoch, all three miners agreeing | e0 v2 `8f8806638d59850f`, e1 v2 `fd9562df32a68313`, e2 v2 `1ae6d90ab299154c`, e3 v3 `5d0dedd9fd9e29a1`, e4 v3 `e81808dcdb02ce05`, e5 v3 `06aff9c1d33e7a13`; no v2 id reappears under v3 |
| Rejected blocks | miners 0 / 0 / 0 (97, 103, 104 accepted); nodes 0 / 0 / 0 `PoW rejected` lines |
| Forks | sinks `082fd39ba65df2ff` on all three nodes, block counts 304 / 304 / 304, one tip each |
| Program and cache ready, one CPU core | a new `(day, class)` cache: v2 epoch 0 219 to 235 ms, the first v3 epoch 177 to 185 ms (two per miner: the 24-minute day rolled at DAA 190); a program swap inside a day 2 ms |
The mixer x4 build-time number the rollout asks for is not visible here: the CPU miner derives dataset words on demand from the cache (no dataset build), so the x4 cost lands on the GPU workers' dataset build, measured by the ca2-mixer playbooks on the PCs.
### Result, run 2 (5 October 2026, 21:46:36Z to 21:51:31Z): the composed class
Binaries rebuilt on ca2-v3 b105a55 (era layout merged on the mixer: `V3_CLASS` = MX4 with the era drawn inside `generate_from_seed_bytes_program_class`, hot `None`), fork 79bd8e10 unchanged. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2146Z-era-mx4.json`. PASS: every check true.
| Check | Measured |
|---|---|
| Switch lines, digest | 3 of 3, the same line as run 1; digest `0186df7d0834d054...` |
| Template class per epoch | epochs 0 to 2 class 2, 3 to 5 class 3; the switch at DAA 180, 171.0 s wall |
| Blocks before / after the boundary | 181 / 124 (selected chain 180 / 122), 305 in all; 102, 102, 100 accepted per miner |
| Program id per epoch, all three miners agreeing | e0 v2 `8f8806638d59850f`, e1 v2 `bb8dd9ddbf9eb63f`, e2 v2 `8ee7a9f33d418e48`, e3 v3 `2d278041ba482dba`, e4 v3 `2ae786d294a8a59d`, e5 v3 `bc36813df2f41b5f` |
| Rejected blocks | miners 0 / 0 / 0, nodes 0 / 0 / 0 |
| Forks | sinks `712c1b212091dcdc` on all three nodes, block counts 303 / 303 / 303, one tip each |
| Program and cache ready, one CPU core | v2 epoch 0 178 ms, the first v3 epoch 181 ms, an in-day swap 2 ms |
CPU hash rate across the switch, run 2, miner cpu0 (one thread, the Mac shared with other agents' builds, so approximate): the cumulative rate read 0.024 MH/s through the v2 epochs (30 to 151 s), then fell to 0.021 MH/s cumulative by 271 s (100 s under v3), which puts the v3 interval rate near 0.017 MH/s, about 30 percent under v2 on the CPU interpreter (the era's strided windowed loads and the mixer path). The node has no per-block verify timing line; the CPU verifier is measured in `igneum-pow` (the mixer agent, one M5 Max core, ms per 32-lane unit, same minute, cited from ca2-mixer 1ab8b21's message of 5 October 2026 22:10Z):
| Path | readwidth | ca2-v3 88dafbc (before the fix) | ca2-v3 d233fa1 (after) |
|---|---|---|---|
| v2 (the live devnet) | 0.607 / 0.610 | 1.332 | 0.609 / 0.611 |
| v3 at x4 | | | 1.238 |
| v3 at x8 (the class) | | | 2.077 (worst cold 2.15) |
The regression was `memhard::derive_items` at m = 1 (2.2x); the fix dispatches to an out-of-line `derive_items_mask`, one instance per cache size with the line mask a constant. A v3 block costs the node about 3.4x a v2 block to verify (2.08 against 0.61 ms per unit).
Epoch 0's v2 id is the same in both runs (`8f8806638d59850f`: a v2 program is untouched by the era code, on the chain as in the packs); the v3 ids differ from run 1 because the era draw is now inside the class.
### Result, run 3 (5 October 2026, 22:15:04Z to 22:19:44Z): the final class, x8 with the era
Binaries rebuilt on ca2-v3 d233fa1 (mixer x8, the era drawn inside the class, the verifier fix), fork 89dfcb95. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2215Z-era-mx8.json`. PASS: every check true.
| Check | Measured |
|---|---|
| Switch lines, digest | 3 of 3, the same line; digest `0186df7d0834d054...` |
| Template class per epoch | epochs 0 to 2 class 2, 3 to 5 class 3; the switch seen at DAA 181, 127.9 s wall |
| Blocks before / after the boundary | 182 / 122 (selected chain 180 / 120), 304 in all; 101, 89, 113 accepted per miner |
| Program id per epoch, all three miners agreeing | e0 v2 `8f8806638d59850f`, e1 v2 `e145305446a6b4ce`, e2 v2 `743ad2a3cab0518a`, e3 v3 `a6523b90cff501e3`, e4 v3 `bc811b3c4b8b1ced`, e5 v3 `e784541f19cdebe5` |
| Rejected blocks | miners 0 / 0 / 0, nodes 0 / 0 / 0 |
| Forks | sinks `a9ce45df8beeaf13` on all three nodes, block counts 303 / 303 / 303, one tip each |
| Program and cache ready, one CPU core | v2 epoch 0 179 ms, the first v3 epoch 191 ms (x8 construction: the cache fill is unchanged, the items are derived on demand), an in-day swap 2 ms |
Epoch 0's v2 id `8f8806638d59850f` is the same in all three runs. Three runs, three v3 classes (mixer x4; x4 with the era; x8 with the era), the same chain behaviour each time: the switch rounds up to epoch 3, no block rejected, one chain.
### Result, run 4 (5 October 2026, 22:25:22Z to 22:31:28Z): a real Metal miner across the boundary (gate G4b)
The fleet-outage case: the three runs above used CPU miners, so the Metal worker's class v3 path had not mined. Run 4 puts node 0's miner on the Metal worker the way the app does (`igneum-miner --worker igneum-bench --prepare-packs <dir> --exit-on-seed-change`, `igneum-bench` built from ca2-v3 00c55aa: a class v3 program comes from the pack's `program_bound.metal`, its day from the pack's `memhard.metal` (the x8 construction, the era layout), keyed by (day, class, era)), genesis bits 0x1e010000 (2^24 hashes per block), CPU miners on nodes 1 and 2. Summary: `docs/plans/counter-asic-2-gate/class-v3-20261005-2225Z-metal-mx8.json`.
| Check | Measured |
|---|---|
| The chain | 182 / 123 blocks across DAA 180, 0 rejected (miners and nodes), sinks `66a33eca15d33ba2` on all three, 304 / 304 / 304, 3 of 3 switch lines, the switch at DAA 180, 264.5 s wall |
| PREPARE lines | 6 (three class v2, three class v3), each 6 to 7 DAA before its boundary, the v3 ones with the pack directory and `class=v3 era=<hex>` |
| The worker's `prepared` for the v3 packs | three: `prepared <seed> <day> 252.5 program 55.5 dataset 196.9 race 0.0 variant base class v3 loads/hash 128 cache-fill 0.9 build 41.1` (the first, with the day built from the pack: 0.9 ms cache fill, 41.1 ms build on the GPU), then 51.1 and 46.0 ms (the day resident) |
| The swap at the boundary | `SEED CHANGE at daa 180 ... (epoch 3): swapped with no pause (prepared 4 s ago, prepare took 252 ms)`; the same at 240 and 300 |
| Blocks on v3 | 124 accepted after the switch (301 in the run), every one re-checked on the CPU: `mismatched 0`; `need` 0; no class or era mismatch line; no refusal, no exit 42 or 44 |
One line before the switch, on the v2 path: `PACK OUT OF DATE: the prepared pair b7eda892... is not the node's epoch bc056261...; rebuilding the program pack` at DAA 60: the epoch-1 seed reported at `boundary - lead` flipped (the quarter-lead confirm is 3 DAA on the 60x profile, 150 on the devnet) and the miner's rebuild path of 05ef0fa3 wrote the right pack. The v2 prepares answered 33 to 35 s after they were sent (the Metal variant race runs inside the prepare, longer than the 10-DAA lead of the profile; 600 s on the devnet), so epochs 1 and 2 compiled inline; a pack program races nothing and answered in 252 ms. The v3 path is clean; the script now judges the Metal checks from the first v3 prepare on (the run's own report flagged the v2 line and read FAIL on that one check).
The PC 2 suite jobs for the same fork (79bd8e10), the G6 evidence (the coordinator's status file carries the SUMMARY lines):
| Job | Tree | What | Result |
|---|---|---|---|
| `build-20261005-215219` | main b105a55 | the Linux node, the six node suites, the app tests | Linux build and app tests passed; `kaspa-consensus` failed on the known flake (`ban_is_decided_by_the_carrying_block`, `UnexpectedDifficulty` in `mine_on_all`, the 0.3.10 cut's section 11 case) |
| `build-20261005-215712` | main b105a55 | `kaspa-consensus` alone | 97 passed, 0 failed, 3 ignored, `ban_is_decided ...` ok, 21:59:45Z |
| `build-20261005-220351` | main 8ea6740 | the other five crates (`kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows`) and the app tests | 22:04 to 22:06:55Z: igneum-exec 17, igneum-miner 18, consensus-core and p2p-flows green, the app 78 + 26 + 8; `kaspa-pow` FAILED on `program_class_v3_seeds_hash_their_own_program_over_their_own_cache` (the fork test compared a chain v3 program's class to `V3_CLASS` with `era: None`; on the era-merged crate the class carries the drawn era inside). Fixed on the fork at 89dfcb95 (the class minus its era is `V3_CLASS`, the era is drawn, another era seed keeps the program id and hashes another program over the same day cache). Re-run: job 4 |
| `build-20261005-221237` | main d233fa1, fork 89dfcb95 | the five crates and the app | 22:12:37 to 22:15:44Z (185 s): every stage ok; kaspa-consensus-core 108 (2 ignored), igneum-exec 17, kaspa-pow 33 + 14 (the v3 engine test, with the igneum-pow feature), igneum-miner 18, kaspa-p2p-flows 7, igneum-app 78 + 26 + 8; 0 failed. With job 2, G6 is green |
The PC's test stage builds the fork's test binaries with the plan's feature set, so the v3 engine test DID run on PC 2 (the earlier reading that it was Mac-only is withdrawn): the PC job is the G6 evidence.
## 6. Tests
| Where | What | State |
|---|---|---|
| igneum-pow `cargo test --release` | 42 unit + 11 pack tests, including `program_classes`, `program_class_and_era_are_checked`, the pinned-pack diffs | pass (Mac, 5 Oct 2026) |
| `proto-cuda/nvrtc/emu/packfile-test.sh` | 13 checks: the attempt rule, the generator rule, a v3 pack with its era, the token matcher | pass (Mac) |
| `host.c`, `worker.cpp` (emulation), `main.swift` | syntax / compile | pass (Mac) |
| fork `cargo test -p kaspa-pow --features igneum-pow` | 14 engine tests including `program_class_v3_seeds_hash_their_own_program_over_their_own_cache` (rewritten at 89dfcb95 for the era-in-class rule) | pass (Mac, 89dfcb95 against d233fa1; PC 2 job 4: 33 + 14) |
| fork `cargo test -p kaspa-consensus-core` | 108 + 7: the switch rounding, the era clock, the params, digest and fast-time file tests | pass (Mac, fork 79bd8e10) |
| PC 2 `build-job.mjs` suites | `kaspa-consensus-core igneum-exec kaspa-pow kaspa-consensus igneum-miner` + `igneum-app` | the coordinator publishes on its go |
## 6a. The integration merges for the ship (5 October 2026, 22:35Z to 22:45Z)
| Merge | Commit | Conflicts, and how they were kept |
|---|---|---|
| readwidth 30ff674 (the OpenCL `__local` declaration rule, the read-width decision record, the per-watt rows) | 3566afd | `emit.rs` (2 hunks: the hot-table kernel arguments, HEAD's superset), `packbench.swift` (readwidth's resident-footprint lines added to HEAD's hot-aware RESULT line), `bench-log.md` (both entries), `read-width.md` add/add (readwidth's version, a pure superset of HEAD's) |
| origin/master 1f0d62c (the 0.3.10 merge) | 49c7e78 | `host.c` (one struct hunk: both fields, `dupOf` and `pci[32]`; master's topology, duplicate-platform and read-back code and ca2-v3's class/era tokens, `mh_word`, hot and mixer fields all auto-merged; `cc -fsyntax-only` clean), `bench-log.md` (both entries) |
Checks after the merges (49c7e78, 22:50Z): the igneum-pow suite 53 + 4 + 19 + 7 pass; the packfile test 0 failures; the CUDA emulation `emu/test.sh` PASS (ready + prepare 1, 64 + 64 + 32 found on pack A, pack B prepared with its self-test, the swap, the self-heal rebuild of pack A, 17 sampled hashes equal to `igneum-pow hash-bound`, the NVRTC source check PASS for 6 files); `proto-opencl/test-generic.sh` on Apple OpenCL PASS (the same protocol, job 5 refused after the swap, 15 sampled hashes equal); the fork's `kaspa-pow`, `igneum-miner`, `kaspad` check clean. The emulation needs `IGNEUM_CUDA_INC` pointed at a checkout's `proto-cuda/nvrtc/redist/include` (the worktree has none).
## 7. Unverified, and what is owed
- The era walk for era >= 1 has never run (the devnet is 180 days from era 1); a pruning-proof sync past era 0 fails with `MissingEraSeed` until an era witness exists in the proof format.
- The Metal worker takes class v3 from the pack (program and day) and mined across the boundary in run 4; the app must pass `--prepare-packs` to its Metal worker (the app branch fixes `engine.rs`, which passed it to the non-Metal workers only).
- The GPU workers' class refusal was checked by the C test of `pf_pack_class_ok` and the syntax of both hosts, not by a live worker on a v3 pack: the integration's bit-exact gate (G1) is where a real worker first builds a v3 pack.
- `next_pair` keeps the current era seed for the next epoch; an epoch boundary that is also an era boundary (once per 180 days) would prepare the wrong era, and the job line then names the right one, so the worker refuses the prepared pair and the miner prepares again (one wasted compile, no wrong block). The VDF era seed replaces the stand-in before this matters.
- Done 22:05Z: ca2-cache rebased as 1950661 fast-forwarded (the first attempt, 2de19e5 on 464d6e1, conflicted in 8 files and was aborted); igneum-pow 53 + 19 tests and the packfile test pass; the fork's kaspa-pow, miner and kaspad check clean against the merged crate (seam unchanged). `V3_CLASS = { era: None, hot: None, ..LoadClass::MX4 }`.
- Done 22:12Z: ca2-mixer 16dfd1e (MX8 candidate, tests) merged at 4e733bb with two one-line fixes the merge needed (3a7fba7 `MX8` gets `era: None, hot: None`; 795472e tests/scratch.rs's `Instr` literals get `win: 0, off: 0`); then ca2-mixer 1ab8b21 (the verifier fix, `V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }`: class v3 is x8 by the coordinator's decision of 22:05Z) merged at d233fa1. Checks on d233fa1: igneum-pow 53 + 4 + 19 + 7, packfile 0 failures, the fork's `cargo test -p kaspa-pow --features igneum-pow` 14 pass, the release rebuild 1 min 57 s.
- Owed (0.3.12, coordinator's ask of 5 October 2026 22:20Z): per-day dataset reuse in the CUDA and OpenCL workers (a `Day` object shared by consecutive pairs, the cache freed after the build); the Metal worker already keys datasets by day. Until then the iGPU tier mines v3 with a dataset rebuild per epoch on those two workers.
- Wire compatibility: `RpcPowEpochInfo` gained five fields in its Borsh form (wRPC) and five proto fields (gRPC); the gRPC side reads an old node's zeros as v2 / never / none; the Borsh form is versioned by `GetBlockTemplateResponse` (version 2 carries the whole struct), so a 0.3.11 wRPC client against a 0.3.10 node reads short: the miner uses gRPC, the console reads JSON (serde defaults).

View file

@ -0,0 +1,49 @@
# Counter ASIC: the public description in four levels
the project lead, 5 October 2026 (night): "not an information overload". Four levels; the layer names appear only from level 3 down, next to their numbers. Numbers come from the final table of `docs/plans/counter-asic-2-status.md`; a number still owed is marked `[owed: ...]`, never guessed. the project lead's copy law throughout.
## Level 1: one sentence (site hero, litepaper abstract)
Built for graphics cards. A custom chip gains under 2x, and the model is public. (The bounty is named only once it is escrowed: docs/plans/funding.md rule 3; D11 for the project lead.)
## Level 2: one site card, one short litepaper section
Three ideas, no layer names, no widths, no SRAM.
**The hash rewrites itself.** A new program every hour, drawn from the chain. Its memory pattern changes with it. The rules change on a schedule fixed at launch. No release, no vote.
**It waits on memory, not maths.** Every hash is a chain of random reads into a table too big for a chip to carry. The wait is the same physics for everyone.
**Miners hold the switch.** Spare defences are written into the rules, switched off. A 90% miner signal turns one on. No fork.
A custom chip gains under 2x. Model published; a bounty follows the external review. [link: the numbers page]
Litepaper only, a fourth paragraph: No hash has stayed free of chips forever. Igneum does not claim to. It claims the gain is small, the response takes a week, and both are measured.
Site card placement: the Mine section of `site/index.html` beside "no chip can be built for it" (which this card replaces: the claim is a bounded gain, not impossibility). Litepaper placement: `site/litepaper.html` section `mining`, replacing the paragraph that begins "Everything above is automatic" and the "What Igneum does not claim" line on chips; the vs RandomX table keeps its rows, with the "Changes over time" row's Igneum cell reading "A new program every hour, its memory pattern and read widths with it; era draws and reserved families on a schedule fixed at genesis".
## Level 3: the numbers page (`site/bench.html`, section "Counter ASIC")
Headline of the chip model (5 October 2026, night): the strongest chip holds the whole 256 MiB cache on-die (about 128 mm^2 and $46 of silicon at N5 by shipped cache-die density, approximate) and computes dataset items on the fly; its gain over the RTX 5090 is 2.4x as the parameters stand, and no write-scratch share within an 8 GB card's budget changes that. The lever that does is the dataset item's mixer cost (x4: 1.8x with a 3x fixed-function factor, verifier 1.6 to 4.8 ms per warp). Decided 5 October 2026 (delegated): the mixer x4 and the cache growth rule enter class v3, so the headline row is the on-die-cache chip against v3 with everything combined. [owed: the combined row from docs/analysis/chip-model-v3.md; if it reads 1.8x, the claim is "under 2x" with the margin stated as thin, and the next levers are named: the mixer x8 and the hot table.]
Per card, the bench table: the v2 class and the v3 class, hash rate, bytes per hash, the latency-bound share (rate over the card's random-read ceiling per load), the CPU verifier per warp, with machine, date and command. The chip model before and after Counter ASIC 2.0 (the m16 model's gain arithmetic at the v2 class and at the v3 class, with the SRAM a mirror needs, cited or approximate as the analysis says). The bounty terms (spec O-1.17: the leaderboard by card model, the standing bounty for any chip design beating a GPU by more than 2x, January 2027). Here the layers are named next to their numbers: read width, per-program mix, scratch, era layout, working set, hot table, cache schedule, the reserved integer-matrix family.
| Card | v2 MH/s | v3 MH/s (era packs, six eras) | Bytes per hash | Latency-bound share | Verifier ms per warp (v2 / v3, one loaded M5 Max core) | Daily 1 GiB build (v2 / v3) |
|---|---|---|---|---|---|---|
| Apple M5 Max (Metal) | 27.68 | 27.85 to 27.98 (spread 0.5%, the final class) | 512 | 1.06 | 0.61 / 2.08 (3.4x; worst cold 2.15) | 21 / 21 ms |
| RTX 5090 (CUDA) | 137.2 | 135.90 to 137.70 (spread 1.3%, the final class) | 512 | 1.01 | the same verifier | 25 / 23 ms |
| RX 9070 XT (OpenCL) | 18.09 | 18.59 to 19.18 (spread 3.1%, the final class) | 512 | 0.95 | the same verifier | 74 / 75 ms |
Every number measured 5 October 2026 (`docs/plans/era-layout.md`, `docs/plans/mixer-x4.md`, `docs/bench-log.md`); the v3 verifier figure is the x8 mixer on one core at load average 5.5 (the fixed crate; the same session matched readwidth's quiet v2 figure within 1%). Bit-exact: every v3 pack's fingerprint equal on the three vendors.
Chip model, before and after (`docs/analysis/chip-model-v3.md`): the on-die-cache recompute chip (the whole 256 MiB cache in SRAM, about 128 mm^2 and $46 of silicon at N5 by shipped cache-die density, approximate) against the RTX 5090's measured 136.1 MH/s at 50 T integer op/s: class v2 333 MH/s, 2.4x; class v3 (mixer x8) 41.7 MH/s, 0.31x bare, 0.92x with a 3x fixed-function allowance (approximate), 0.76x at equal silicon. The claim "under 2x" holds with the margin stated: 8% on the allowance (a 3.3x allowance reads 1.0x), 9% on the budget. Next levers, named: the mixer at x16 (the verifier at about 4 ms per warp, inside the 10 ms gate; a 2019-class core unmeasured), a hot table small enough to stay resident beside the streaming dataset (measured and not adopted tonight: 32 to 96 MiB tables cost the GPU 7 to 20% and help the chip).
Levers measured and not adopted (5 October 2026): wider reads (16 and 64 B: no card gains, the 5090 goes bandwidth-bound at 64 B), the per-load width mix (5.5 to 22.3% spread), a per-warp write scratch (the chip keeps it implicitly: 2.4x at every share), the hot table (above). Reserved, switched off: the integer matrix family R1 (mm8, native on all three vendors as a tile; dp4a 1.17x a step on the 5090, 1.06x on the 9070 XT, emulation 1.6x on Apple) and the epoch length (600 s to 2 hours by 90% signal; a per-program FPGA bitstream mines 0% of a 600-s epoch).
AMD RDNA 4 sits at about a seventh of a 5090 on this hash by its dependent-read rate (2.4 G against 17.5 G reads per second), 2.2x worse per pound at list prices and 4.9x worse per watt (read-width.md section 4.1, approximate); the card's memory system, not a tuning gap.
Chip model, before and after: [owed: from docs/analysis/sram-mirror.md after the shipped-density correction (256 MiB on-die at about 130 to 165 mm^2 by AMD 3D V-Cache and TSMC N5 macro density, 54 to 83 mm^2 bit-cell-only lower bound), the hot-table and scratch analyses; the on-die-cache recompute chip is a named row per variant].
## Level 4: the analysis documents
`docs/plans/counter-asic-2.md` (the plan and the layer table), `docs/plans/read-width.md`, `docs/plans/era-layout.md`, `docs/plans/hot-table.md`, `docs/analysis/scratch-soundness.md`, `docs/analysis/sram-mirror.md`, `docs/analysis/int8-matrix-family.md`, `docs/analysis/m16-recompute-attacker-2026-10-05.md`, the bench log entries of 5 October 2026 (night), the specification sections 1.4 to 1.13.

View file

@ -0,0 +1,131 @@
# Generator class v3 on the live devnet: rollout plan (5 October 2026, night)
Scope: the DEVNET only. the project lead delegated the three decisions for the devnet before going to bed (5 October 2026, about 20:10 UTC, through the coordinator): "Counter ASIC 2.0 fully deployed" tonight. The public testnet is not open; its genesis takes v3 from day one. The devnet is ours and a reset is acceptable.
Shape and rules follow `docs/plans/finality-v3-rollout-devnet.md` and the publish record `docs/plans/finality-v3-devnet-publish.md`: one height switch read from the override file, every node carries the same object before the height, the PCs get igneumd only through an OTA app version, the activation height leaves at least three hours from the manifest publish. Nothing in this file has run on the devnet. The numbers marked `<...>` are filled by the integration branch `ca2-v3` and the decisions of section 6; the plan is published with them, not before.
## 1. What changes and what does not
Only the lottery hash's program class changes, and only from the first epoch at or above the height. One switch, `program_class_v3_activation_daa`, in `Params` and `OverrideParams` like `difficulty_v2_activation_daa`; default `u64::MAX` (never) on every network. Because one epoch has one program (spec 01 section 1.12), the switch keys on the EPOCH: epoch `e` is class v3 when `3,600 e >= N4`, so the activation is rounded up to an epoch boundary and a block's class is a function of its DAA score alone, as today.
Class v3 = generator version 3: the width rule `<W>` (layer 1 or 2, decision 1), no scratch (layer 3 decided out: scratch share 0), the era draw of the table layout and the working set (layers 4 and 8), the hot table of `<S>` MB from the epoch seed (layer 5), the cache growth rule of layer 6 (option C) and the M16 mixer x4 in the dataset item construction (decided 5 October 2026, delegated). New program id (`generator = 3` in the id's preimage, spec 1.4.6), new packs and vectors, new `IGNEUM_GENERATOR` in every pack, a pack of the other version refused by every implementation (spec 1.4.5 already says so).
What does not change: the chain, the genesis, the databases, the day key and the 256 MiB cache fill, the dataset items (spec 1.8.5, if the era interleave keeps the item values; the era-layout document says what it costs otherwise), finality, fees, proving. The SP1 guest does not read the lottery hash (`proving/igneum-prove` has no dependency on `igneum-pow`; the pinned guest of DAA 210,000 is a fee-table switch), so no new guest is pinned. The node's `EpochSeeds` gains the class and the era bytes; `IgneumEngine::epoch_for` builds the v3 `Epoch` from them; the miner's `export-pack` and the serve protocol's job line carry the class so a GPU worker regenerates the right pack from the seed bytes.
The consensus digest (`Params::consensus_digest`, ledger X18) covers every activation height, so the new field enters the digest and every node must carry the same object before any node reaches the height; a node without the field is refused at the handshake once the others carry it, which is the protection the digest exists for. Binary rollout first (the digest flips when the binary carries the field at `never`), the height second.
## 2. The binaries
Built from `<node branch>` at `<commit>` on the PCs through `tools/build-job.mjs` (standing rule 5 October 2026), the Mac binary under the build lock. The table is filled at build time: platform, path, sha256, how it was verified (`--version`, the switch's first line on a private suffix, `strings` carries the field name).
## 3. The activation height N4, and how every node learns it
`N4 = DAA at the manifest publish + 10,800` at least, chosen as `DAA now + 14,400` rounded up to the next epoch boundary (a multiple of 3,600), checked at publish (`N4 - DAA >= 10,800`). The packaged line carries every switch:
```
NODE_OVERRIDE_PARAMS='{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":N4,"proving_v1_activation_daa":N5,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000}'
```
The same nine-field object goes verbatim into the override files of Mac node 1, the observer node and the seed, and into the manifest's `consensus.override` (`publish-manifest.sh --override`). N5 = DAA at publish + 14,400 (the proving v1 switch, no rounding). Two digests, read on the 0.3.11 Mac node (igneumd bd7f043c..., fork 89dfcb95, 22:5x UTC): with no override file c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c (the rolling-upgrade digest, equal to the node agent's pinned test); with the nine-field object at N4 = N5 = 154,800 the ACTIVATION digest 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888, the value every node must print after the publish; the node logs "Program class v3 ... active from epoch 43 (DAA 154800 ... epochs of 3600)" and "Proving v1 ... paid from DAA 154800, 8 blocks a segment, unproven after 600 DAA, aggregator share 1000 bps". Binaries from release-0.3.11 23bc2b2: Igneum-Miner-0.3.11.dmg b7e81d4f6f3af9f9179e29faa72c844b795df757dfb2e4cd1d56cd454e78e1e7 (41,592,041 bytes; the packaged json carries the nine fields, read back from the image); the seed's Linux igneumd 63cf490d... (glibc 2.34, 89dfcb95 inside). The era seed for the devnet: the stand-in of `docs/plans/era-layout.md` (the hash of the last selected-chain block below `15,552,000 n - 7,200`; era 0 on the devnet uses the genesis hash), until the 1-hour VDF of spec 4.4 is in the node.
## 4. The order
1. The digest flip: a node build that carries `program_class_v3_activation_daa` at never on every node (hand nodes and the seed first: `infra/devnet/restart-hand-nodes.sh '<object without the new field>'`, then the app version through the manifest; every node prints the same `Consensus params digest`).
2. Fix N4, cut the app version (`packaging/mac/packaged-config.sh`, the three version files), commit as igneum-labs.
3. The Windows payload inputs (`packaging/windows/push-inputs.sh`) and the Mac DMG (`packaging/mac/build-dmg.sh`), the manifest (`publish-manifest.sh --activation-height N4 --deadline-note "program class v3" --override '<object>' --deploy`), `fetch-ci-artifacts.sh --deploy`.
4. The observer, the seed, Mac node 1 with the object; each node's first lines show every switch and `Program class v3 from the override file: active from epoch <N4 / 3,600>`.
5. HiveOS: `packaging/hive/make-hive-package.sh` republished with the v3 `igneum-miner` and workers, same version string as the apps.
6. The watch: before `N4 - 1,800` both PCs on the new app version (STATUS lines); at the boundary every miner's `prepare` of the v3 pack (the hot-swap entry's shape) and the first v3 block's program id on the observer; the hash rate per card against the measured v3 numbers of `docs/plans/counter-asic-2.md`'s table; zero `pack refused` lines; the CPU verifier time per block in the node log against the measured ms per warp.
### 4a. Two rules for the publish and every PC job (C32)
- The 0.3.11 update-now goes to PC 2 only after the prover-floor agent's server build (floor-build-3, under /opt/igneum-floor in WSL2, published 22:27 UTC, 25 to 90 minutes) has closed: an app restart ends the running job. The update-now takes a machine list (as 0.3.10's did): the Mac, the laptop and PC 1 first, PC 2 last.
- Re-fetch after every app update: an app update clears the jobs folder (the 0.3.10 install at 21:49 UTC took PC 1's AMD kit with it), so every fetch-then-run pair re-publishes its fetch after an update, and every run playbook opens with a presence check of its kit that fails with "kit missing: republish the fetch after the app update".
- No PC job raises an elevation prompt on either PC for the rest of the night (C35): both unexplained app quits tonight came 20 to 41 s after an administrator prompt beside the running installed app (PC 2 20:00:49 to 20:01:09Z, PC 1 22:30:25 to 22:31:06Z), the engine has no self-relaunch after a quit, and nobody is at a keyboard; the two-minute test of that class (raise one prompt from a job while the app mines, cancel it, read the quit line, which ember-tune b671c8b stamps with its source) runs in the morning with the project lead present, or tonight only after the relay relaunch path is proven and the rollout is done. The sweeps and Ember's run stay held; the floor, aggregation-cost and M16 jobs raise no prompt.
## 5. Rollback
Before N4: remove the field on every node and restart; nothing has happened (the digest flips back, so every node at once). After N4: there is no rollback by restart, because blocks mined under v3 verify only under v3. A rollback is a second height switch back to v2 at a later epoch, carried the same way. This is why the measurements of the plan come first.
## 6. The decisions, by the project lead's rules (devnet)
the project lead's rules, applied by the coordinator and recorded here with the number that decided each:
| Decision | the project lead's rule | Choice | The number |
|---|---|---|---|
| Width (layer 1) | the widest read that keeps every card we own latency-bound (achieved loads within 90% of the probe ceiling) with margin on the 5090 (its bytes per hash under a third of its bandwidth at the measured rate) | DECIDED (5 October 2026, delegated): keep v2, 128 x 4 B. w16 passes the rule (shares 0.90 / 0.84 / 1.03, 18% of the 5090's stream) but closes nothing and does not move the chip row; w64 and w64x4 make the 5090 bandwidth-bound (share 0.58 / 0.56, 37% of stream) | `docs/plans/read-width.md` (readwidth e752fc7): v2 5090 136.1 MH/s, 9070 XT 18.15, M5 Max 27.74 (gap 7.5x); w16 139.8 / 17.90 / 28.26 (gap 7.8x); w64 71.9 / 17.59 / 28.27 (gap 4.1x); the 9070 XT does 2.4 G dependent reads/s at every width |
| Per-load mix (layer 2) | in, if the min-to-max spread across six programs is under 5% per card | DECIDED (5 October 2026, delegated): out | spreads of the median over six programs: mix 50/35/15 5090 18.8%, 9070 XT 7.4%, M5 Max 11.3%; mix 25/50/25 22.3% / 5.5% / 8.1% |
| Scratch share (layer 3) | the smallest share at which the chip model's gain falls under 1.5x at the lowest GPU cost, within the 6 GB working-set cap | DECIDED (5 October 2026, delegated): 0. No share under the cap moves the on-die-cache recompute chip, so layer 3 is not adopted into v3 | `docs/analysis/scratch-soundness.md` (ca2-soundness a465881): chip 333 MH/s against the 5090's measured 139.7 = 2.4x at 0% RMW; 2.4x at 12.5 / 25 / 50% replaced (chip 381 / 443 / 661 against 160 / 186 / 279 projected) and 2.4x or more added, at 32 and 128 KB; the chip keeps the scratch implicitly in 80 to 320 B per lane because the verifier resets it per unit |
| Activation height N4 | devnet tip + 14,400 at publish, checked >= 10,800 at publish, rounded up to the epoch boundary | `<at publish>` | |
### 6a. The chip model's headline, and how the scratch share is chosen
The layer 6 finding changes the headline: the strongest chip holds the whole cache on-die (about 130 mm^2 at N5/N3E and 165 mm^2 at 7 nm by shipped-product SRAM density, AMD 3D V-Cache 1.56 MB/mm^2 and TSMC N5 HD macros; 54 to 83 mm^2 is the bit-cell-only lower bound; `docs/analysis/sram-mirror.md` after the 20:18 correction) and computes dataset items on the fly through M16's mixer. The before-and-after table must carry that chip as a named row against the GPU for each variant, and the scratch share is chosen by that row: the smallest share at which the on-die-cache chip's gain falls under 1.5x. Result (20:23 UTC): no scratch share under the 6 GB cap gets that chip under 2x; the gain is 2.4x at every share. The site's "under 2x" claim is therefore qualified until the mixer multiplier or the cache rule closes it: M16's mixer multiplier x2 gives 1.2x (3.6x with a 3x fixed-function factor) at 0.8 to 2.4 ms verify per warp, x4 gives 0.6x (1.8x with the factor) at 1.6 to 4.8 ms, inside the 10 ms gate, with the 5090's daily dataset build at 27 and 54 ms. DECIDED (5 October 2026, delegated under "execute the full 2.0 plan" and "deploy what is absolute best"; the project lead confirms for the public testnet genesis): M16 mixer x4 goes into v3 behind the same activation. Numbers: attacker 0.083 Ghash/s at 50 T op/s (0.36x bare against 229 MH/s; 1.8x against the 5090's measured 139.7 MH/s with a 3x fixed-function factor, approximate); verifier 1.6 to 4.8 ms per warp (inside the 10 ms gate); the 5090's daily dataset build 54 ms (13.4 x 4, measurement owed). Layer 3 stays out (scratch share 0); its soundness document and pack-contract tests are kept because the construct is sound and may return. The chip model's headline row becomes the on-die-cache recompute chip against v3 with everything combined (x4 mixer, the hot table, the width rule, the era draws, the cache growth), fixed-function factor included; the public level 3 shows that row: if it reads 1.8x the claim is "under 2x" with the margin stated as thin and the mixer x8 and the hot table named as the next levers. The dataset and cache vectors are re-cut once for v3 (cache growth rule and x4 together), the soundness suite re-run on the new construction, bit-exact on the three cards, the verifier per-warp time measured on the Mac; the daily dataset build time quoted for the 5090, the Mac and the 9070 XT. Branch ca2-mixer carries it. Refinement (coordinator, 21:28 UTC, delegated under "as strong as the measurements allow"): x8 is built and measured beside x4 on the same packs (verifier per warp on one Mac core, the 1 GiB daily build on the 5090, the M5 Max and the 9070 XT or gfx1036, the chip row at equal silicon with the 3x factor); x8 goes into v3 if the per-warp verify stays under 10 ms on one core and the daily build stays under 1 s on every card we own, otherwise x4 with the thin margin stated in level 3 and x8 named as the next lever. DECIDED (5 October 2026, 22:06 UTC, delegated): x8. Both halves pass: the per-warp verify at x8 is 2.79 ms on a loaded M5 Max core (2.1x v2; about 1.3 ms quiet, approximate) against the 10 ms gate; the daily 1 GiB build does not move with the mixer on any discrete card (RTX 5090 23 to 25 ms, RX 9070 XT 72 to 77 ms, M5 Max 21 ms at x1, x4 and x8: latency-bound), 13x to 40x under the 1 s bar (PC 1 jobs fetch-mixer-x4-20261005 and run-mixer-x4-pc1-20261005, 22:00:08 to 22:04:39Z, both cards restored, every pack's fingerprint equal to the Mac's). V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }. Chip row at x8: 1,198,080 ops per hash, 41.7 MH/s at 50 T op/s, 0.31x bare, 0.92x with the 3x fixed-function factor, 0.76x at equal silicon: the claim reads under 1x with the factor, margin 8% on the factor and 9% on the budget (chip-model-v3.md). Verifier on the fixed crate (ca2-mixer 1ab8b21, 22:07 UTC, same input beside readwidth's binary, load 5.5): v2 0.609 / 0.611 ms per unit (readwidth 0.607 / 0.610), x4 1.238 / 1.237 (2.0x, worst cold 1.40), x8 2.077 / 2.058 (3.4x, worst cold 2.15): 4.8x inside the 10 ms gate on this loaded core. The vectors are re-cut once on this class. Cost: pool shares per core and IBD time scale with the verifier (x8: 2.1x v2); the integrated tier mines v3 with a restart per epoch until per-day dataset reuse lands (0.3.12). The scratch-soundness analysis carries this table (`docs/analysis/scratch-soundness.md`, question 2).
### 6b. The user tiers (the consequences rule)
AMD RDNA 4 (the RX 9070 XT) sits at about a seventh of an RTX 5090 on this hash by its dependent-read rate (2.4 G against 17.5 G reads per second at 1 GiB, measured tonight), 2.2x worse per pound at list prices (0.032 against 0.072 MH/s per pound, approximate) and 4.9x worse per watt (read-width.md section 4.1). This is the card's memory system, not a tuning gap: no read width closes it without making the 5090 bandwidth-bound. The level 3 numbers page states it.
## 7. Gates before any publish (all of them, no exceptions)
| # | Gate | Evidence required | State |
|---|---|---|---|
| G1 | bit-exact v3 on all three vendors against the Mac reference | GREEN on the final class (job run-ca2-era-pc1b-20261005, 22:16 to 22:21Z, exit 0 in 304 s, both cards restored, app 0.3.10, the 9070 XT present as gfx1201): the seven final-class packs' 2^24 fingerprints equal on the RTX 5090 (CUDA/NVRTC), the RX 9070 XT (OpenCL) and the M5 Max (Metal): mx8-devnet-epoch0 90f794dd556f7a3b, era-0 8e8e070db4eea52d, era-1 891c01b8563bb47e, era-2 e54279fed2831b5d, era-3 77e0ba8abbd0ae62, era-4 d898d8f4f2e7684b, era-5 a6927db380f7efb2; self-test PASS on every pack on both cards. RULING (coordinator, 5 October 2026, about 21:10 UTC): the integrated gfx1036 (RDNA 2, AMD OpenCL 3683.0) satisfies the AMD vendor tonight, because G1 is a compiler-and-ISA property and gfx1036 carried the v1 and v2 conformance; the 9070 XT's hash-rate and power rows are owed and taken when its link is back | GREEN on the final class (run-ca2-era-pc1b-20261005, 22:16 to 22:21 UTC) |
| G2 | the CPU verifier exact on 1,000 random hashes per card | GREEN: one serve-mode job of 1,024 nonces at target ff..ff per card and pack (every nonce a found line), re-hashed on the Mac with `igneum-pow hash-bound --prehash 00..01 --count 1024` on the same pack: RTX 5090 era-0 1,024 of 1,024 and mx8-devnet-epoch0 1,024 of 1,024; RX 9070 XT era-0 1,024 of 1,024 and mx8-devnet-epoch0 1,024 of 1,024 (the same job) | GREEN |
| G3 | the generator soundness suite green, the new scratch tests included | `cargo test` in igneum-pow, `tests/packs.rs`, the Metal fuzz, edge, stats, determinism runs on the v3 class | GREEN (the crate suite 53 + 4 + 19 + 7 on ca2-mixer 1ab8b21 and the release tree; the Metal fuzz 200 of 200 and 50 of 50 on x8, the edge, stats and determinism runs; the scratch tests 7 of 7; 22:12 UTC) |
| G4 | the fast-time 3-node network mining across a v3 activation | 0 rejected blocks, 0 forks, every node's first lines show the switch, blocks on both sides of the boundary. Run 1 PASS (21:33 to 21:38 UTC, fork 79bd8e10 + igneum-pow 66eeba3, the mixer-x4 class without era): 3 of 3 nodes print the switch line (active from epoch 3, DAA 150 rounded up to 180 at 60-DAA epochs); templates class 2 for epochs 0 to 2 and class 3 for 3 to 5; 181 blocks before and 124 after DAA 180 (305 total, 3 CPU miners); program ids agree on all 3 miners (e3 v3 5d0dedd9fd9e29a1, e4 e81808dcdb02ce05, e5 06aff9c1d33e7a13); rejected 0/0/0 on miners and nodes; one sink 082fd39ba65df2ff on all three at 304/304/304 blocks; a new (day, class) cache 177 to 235 ms on one core, in-day swap 2 ms. `docs/plans/counter-asic-2-node.md` section 5; summary `docs/plans/counter-asic-2-gate/class-v3-20261005-2133Z-mx4.json`. Run 2 PASS (21:46:36 to 21:51:31 UTC, igneumd and igneum-miner rebuilt on b105a55 = the era and mixer composed class, hot None): every check true; 181 blocks before and 124 after DAA 180; v3 program ids e3 2d278041ba482dba, e4 2ae786d294a8a59d, e5 bc36813df2f41b5f on all three miners (the v2 id for epoch 0 8f8806638d59850f unchanged from run 1: v2 byte-identical on the chain too); rejected 0/0/0; one sink 712c1b212091dcdc at 303/303/303; 3 of 3 switch lines; cache ready v2 178 ms, first v3 181 ms, in-day swap 2 ms Run 3 PASS on the FINAL class (22:15:04 to 22:19:44Z, binaries from ca2-v3 d233fa1 = x8 + era + the verifier fix, fork 89dfcb95): 182 / 122 blocks around DAA 180, 304 in all; v3 ids e3 a6523b90cff501e3, e4 bc811b3c4b8b1ced, e5 e784541f19cdebe5 on all three miners; epoch 0's v2 id 8f8806638d59850f the same in all three runs; rejected 0/0/0; one sink a9ce45df8beeaf13 at 303/303/303; 3 of 3 switch lines; cache ready v2 179 ms, first v3 191 ms. Summary `docs/plans/counter-asic-2-gate/class-v3-20261005-2215Z-era-mx8.json` | GREEN (runs 1, 2 and 3; run 3 on the final class) |
| G5 | the PC-built Windows workers and the Mac workers from the same commit | From release-0.3.11 23bc2b2: igneum-worker-cuda.exe 2b3b8c92885442179f6bf2907c6f3eb453dc4a19908d90fd05981a09b7c2674c (1,536,512 bytes), igneum-worker-opencl.exe edc4a75da3b93d814caa69fd635010780d63d5b622ec24c3741d433c584f91e3 (478,208), both with the resource block, different from 0.3.10's pair; the Mac worker and the DMG from the same tree (the shipper's step report) | GREEN at the workers; the DMG and the node builds in flight |
| G4b | the Mac mines v3: a real Metal miner across a v3 boundary through the miner's `--prepare-packs` flow, and the app passes that flag to the Metal worker | Found 22:21 UTC: `app/igneum-app/src/engine.rs` `miner_args` pushes `--prepare-packs` only when `card.worker != "Metal"` (line 1468 on a223ca9), so the Mac app never hands its Metal worker a prepare pack, and under class v3 the Metal worker compiles v3 only from a prepared pack (servePackProgram): at the first v3 epoch every Mac would answer `need` lines and stop, the 18:23Z outage class. Two closes before the ship, both in hand: (a) the app fix on 0.3.11's app branch (push the flag for every worker with the platform's path separator, a unit test on `miner_args` for a Metal card; assigned to the proving agent on top of a223ca9); (b) gate run 4: the fast-time network with one real Metal miner on the Mac across the activation (the prepare lines on both sides, the worker's `prepared` line for the v3 pack, found or accepted blocks on v3, no `need` or mismatch line; assigned to the node agent). If (a) is not in the app tree at the cut, the ship does not go: a Mac that cannot mine v3 at activation is a fleet outage, and the activation height (tip + 14,400) is not far enough to carry the fix in 0.3.12 safely | GREEN. (b) gate 4 (22:25 to 22:31Z, a real Metal miner on node 0, igneum-bench from ca2-v3 00c55aa): three v3 PREPARE lines with the pack dir and class=v3 era=<hex>; the worker's v3 `prepared` lines (252.5 ms the first: program 55.5, dataset 196.9, cache fill 0.9, build 41.1; then 51 and 46 ms with the day resident); every swap "with no pause"; 124 blocks accepted on v3 (301 in the run), cpu re-check mismatched 0, need 0, no mismatch or refusal, no exit 42 or 44; chain 182 / 123 across DAA 180, 0 rejected, one sink. A second outage found and fixed before the run: the Metal worker's serveDataset built every day with the Swift version 2 construction keyed by day only, so a v3 program would have hashed over an x1 dataset; now a v3 prepare builds the day from the pack's memhard.metal and the store keys datasets by (day, class, era), commit 00c55aa |
| G6 | the node change on a fork branch from the 0.3.10 tip 21d4c73c with suites green on PC 2 | the build job id and its SUMMARY line. State 21:27 UTC: fork ca2-v3-node 79bd8e10 (2e464e81 the class switch + ba43cf0f the proving-v1 merge + 79bd8e10 the digest re-pin); Mac: cargo check of the seven crates clean, kaspa-consensus-core 108 + 7, kaspa-pow with igneum-pow 14 (the v3 engine test included); the PC 2 job publishes at 21:45 from the ca2-v3 worktree (the PC's test stage runs kaspa-pow and kaspa-consensus without the igneum-pow feature, so the v3 engine test's evidence is the Mac run). Expected consensus digest for a scratch devnet node with no override file after the flip: c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c (0.3.11; 0.3.10's is 9409deda...) | Mac green. PC 2 job build-20261005-215219 (21:53:01 to 21:55:48Z, 167 s): the Linux build ok, igneum-app tests 78 + 26 + 8 passed, but kaspa-consensus 96 passed and 1 FAILED: processes::finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list, UnexpectedDifficulty(487112384, 487129578) in mine_on_all: the SAME flake the 0.3.10 cut hit on 21d4c73c under the six-package parallel run (release-0.3.10.md section 11: it passes alone, twice). Treated as that cut did: job 2 of 3 build-20261005-215712 (21:57:12 to 21:59:45Z, 122 s): kaspa-consensus alone 97 passed, 0 failed, 3 ignored in 2.10 s, ban_is_decided ... ok; job 3 of 3 build-20261005-220351 (22:04 to 22:06:55Z, 159 s): igneum-exec 17, igneum-miner 18, consensus-core and p2p-flows green, the app 78 + 26 + 8, but kaspa-pow 13 passed and 1 FAILED: igneum::tests::program_class_v3_seeds_hash_their_own_program_over_their_own_cache (consensus/pow/src/igneum.rs:910) compares a v3 program's class to V3_CLASS with era: None, while the merged crate puts the drawn era inside the class (a stale fork test, not a behaviour fault; the PC's stage does run the v3 engine test, so the PC job is the evidence). The fork test is being fixed; job 4 build-20261005-221237 (published 22:12:37Z: main d233fa1 = era + cache + the mixer fix with V3_CLASS = MX8, fork 89dfcb95) ran the five crates and the app on the final tree: every stage ok (22:12:37 to 22:15:44Z): kaspa-consensus-core 108 (2 ignored), igneum-exec 17, kaspa-pow 33 + 14 (the v3 engine test with the igneum-pow feature), igneum-miner 18, kaspa-p2p-flows 7, igneum-app 78 + 26 + 8, 0 failed. With job 2 (kaspa-consensus alone 97) G6 is GREEN on the final tree | GREEN |
If any gate fails: stop at that gate, write why in `docs/plans/counter-asic-2-status.md`, do not publish.
## 8. The release
0.3.11 through the shipper's pipeline (`tools/ship-app.mjs`, the plan shape of `docs/plans/release-0.3.10.md`). The 0.3.10 shipper finishes 0.3.10 first, then takes the 0.3.11 tree, or the coordinator ships 0.3.11 with the same runbook if the shipper has stopped. The publish order is that of `docs/plans/finality-v3-devnet-publish.md`: the override field, the digest handshake, hand nodes and the seed first, then the manifest with `--activation-height` and the deadline note, then the apps, then the digest sweep, then the HiveOS package republish.
### 6c. Card lifetime: cache residency and the growth mapping (from `docs/analysis/card-lifetime-2026-10-05.md`, merged)
| Decision | Choice | The number |
|---|---|---|
| Cache residency on the GPU | DECIDED (5 October 2026, delegated): the cache is FREED after the daily dataset build; the hash reads the dataset and the hot table only, never the cache. hot-table.md's resident reading is corrected to this | The daily rebuild is the only cost: cache fill 0.67 ms and dataset build 13.4 ms on the RTX 5090 at x1 (bench-log, 3 October 2026), 2 ms and 13 to 30 ms on the M5 Max; at mixer x4 the build is about 54 ms (measurement owed on ca2-mixer). Freeing it moves the 12 GB tier from year 12 to year 28 under mapping (b) |
| Dataset growth mapping (a: continuous 2 + 0.5 GiB a year with a multiply-shift index; b: power-of-two steps at years 4, 12, 28, 60 with `AND MASK`) | RECOMMENDED for the project lead: (b). It keeps `AND MASK` and every vector's size, it is what option C's "doubles when the dataset doubles" already assumes, and it is the only mapping under which the litepaper's "4 GB about four years, 8 GB more than a decade" is true (under (a) a 4 GB card is out within 1 to 1.5 years, an 8 GB card at 6 to 7.5 years) | card-lifetime table 2: 4 GB out at year 4 (b) or 1.0 to 1.5 (a); 8 GB year 12 or 6.3 to 7.5; 12 GB year 28 freed or 12 resident; 24 GB year 60 freed |
Public lines to fix on the integration branch (card-lifetime table 3): `site/index.html` "2 GB, growing" gains the rate ("2 GB at genesis, doubling at years 4, 12 and 28"); "Any 4 GB card" becomes "any 4 GB card at launch, 8 GB from year 4"; the litepaper's "4 GB about four years, 8 GB more than a decade" stays with mapping (b) and gains "under the step schedule"; `docs/evidence.md` gains a row for the card-lifetime claim labelled designed. hot-table.md line 73's 8 GB row is corrected (it counted a 5090's 8,160 warps; a real 8 GB card has 20 to 24 SMs).
## 7b. Hardware events (for the morning summary)
| When (UTC) | Event | What the app did | For the project lead |
|---|---|---|---|
| 5 October, at install (earlier today) | the RX 9070 XT in the Sonnet Breakaway Box 850T5 over USB4 went Code 43 | came back after a driver reinstall and a reboot | |
| 5 October, about 20:40 | the 9070 XT dropped off PC 1's bus: Get-PnpDevice -Class Display lists only the integrated AMD Radeon Graphics (gfx1036) and the RTX 5090; after `pnputil /scan-devices` at 20:45:34Z the card is still absent and the USB4 list shows only the host and root routers: the "USB4 Router (2.0), Sonnet Technologies Breakaway Box 850T5" present at 17:18Z is gone, so the box itself is off the link | the AMD worker (igneum-worker-opencl --device 1) mines the gfx1036 at 3.12 MH/s; the 5090 keeps mining; nobody was woken, PC 1's app was not restarted; a 10-second rescan probe (pnputil /scan-devices, the USB4 router status) was granted | the second eGPU link fault today: reseat the USB4 cable and the eGPU's power; the 0.3.10 hot-plug code shows the card as "removed" and picks it up again without a restart |
| 5 October, 21:22:59 to 22:21 | the card dropped again at 21:22:59 (the third drop), was back and used by the hot-table (21:31 to 21:35), the era (21:46 to 21:51 and 22:16 to 22:21) and the mixer (22:00 to 22:04) jobs as gfx1201, and was gone again by 22:24:09 (the fourth drop; no Sonnet or USB4 router device) | the OpenCL worker falls to the gfx1036 (--device 1) when the card is gone; every measurement that names gfx1201 ran while it was present | the link flaps on a scale of tens of minutes: reseat the USB4 cable and the eGPU's power, try another port or cable; the AMD clock and power sweep is owed on this |
| 5 October, by 21:09 | the 9070 XT is back on PC 1's bus: the reproducible-benchmark package's OpenCL worker listed it as opencl:1 and the app switched it off and on through api/cards; no restart, nobody touched the box | the app's AMD worker returns to it at the next prepare | the link drops and returns by itself; the reseat is still worth doing in the morning |
| 5 October, by 21:22:59 | the third drop: no Sonnet or USB4 Router (2.0) device present, the display list shows only the gfx1036 and the 5090 | the app lists nvidia:0 and amd:1:gfx1036; the 5090 keeps mining at 124.5 MH/s (app log STATUS lines through 21:23:26Z) | the link is flapping: reseat the USB4 cable and the eGPU's power in the morning, and consider a different USB4 port or cable; every AMD measurement tonight runs on the gfx1036 fallback unless the card is present at the job's own probe |
## 7a. A dated constraint from the consequences review (C1)
The fee switch H = 210,000 arrives about 18:45Z on 6 October (DAA 137,041 at 22:32:54Z on 5 October; 1.002 DAA/s averaged since 15:40Z; the 19:50Z in fee-switch-devnet.md is an hour late). The 0.3.10 node's export RPC carries no daaScore and no feesV1ActivationDaa (the proving v1 fork does), so from H every app prover on 0.3.10 has its shards refused and the devnet's proving goes dark. 0.3.11 must be on every prover before 16:00Z on 6 October; if it is not, the fee switch is republished at H = tip + 86,400 by the fee-switch plan's rule (a digest flip, every node in one sweep). The status file carries the timing against H.
## 8a. Proving v1 rides with it
the project lead delegated the proving v1 decisions to its agent (acd4f36bc2c07a4e2). Handoff received 21:05 UTC: fork proving-v1 = ece42979 on 21d4c73c (eb32c645 the feature on protocol 15 / message 75; 2dfad910 the rebased pool test; 3203c8d0 N = 8; ece42979 the digest test edit); Mac unit tests on it: consensus-core 26, exec 8, flows, 0 failed; app proving-v1 FINAL e0de2ab on 5b0d54f (a223ca9 the resume fix, e0de2ab the Metal --prepare-packs fix; docs-only commits between) (6dc686a plus the resume fix: the resume path re-arms every slot without a live worker, re-exports the pack, and logs "<card> is not mining 90 s after resume"; the known-failed case of PC 2's 21:25:11Z resume is the unit test the_pc2_resume_of_21_25_11z_restarts_under_the_new_rule_and_not_the_old, `cargo test -p igneum-app resume` 3 passed; cause: Cmd::Resume re-armed only faulted slots after stop_miners had cleared every restart_at; the tier numbers from the miner-on curve, the AMD and Apple "mines and does not prove" line, the CPU path refused under 32 GB of RAM, the 24 GB tier marked "measured on the 32 GB card", the fast-time file at unproven_daa 10; `cargo test -p igneum-app provedefault` 6 of 6 on the Mac). Harness on the final fork tree: `IGNEUM_PV1_BIN=vendor/igneum-node/target-pv1/release tools/lock/with-lock.sh run node tools/proving-v1/net.mjs --secs 1500 --segment 8` gave "RESULT proving v1 harness: PASSED (21 checks) in 244.4 s" at 20:56:45Z. Override fields at publish: proving_v1_activation_daa = tip + 14,400, proving_v1_segment_blocks 8, proving_v1_unproven_daa 600, proving_v1_aggregator_share_bps 1000; they enter the digest only once the activation is set. Pinned guests unchanged. Mixed fleet: a 0.3.10 node peers with a 0.3.11 node at protocol 14 and never receives message 75; it carries segment records as miner bytes and pays nothing for them; before H the fleet is unchanged, after H only 0.3.11 producers carry and pay segment records and the shard split moves to 90/10, so every node must be on 0.3.11 before H (the fee-switch rule, section 7a). The PC 2 suite job on the merged tree (ca2-v3-node + ece42979) is the suite evidence for both halves. The app half also carries the resume fix (assigned 21:47 UTC to the proving agent on its app branch on top of 6dc686a): engine.rs's resume path restarts every enabled card's worker and the pack export if the pack is stale, re-checks within one tick that every enabled card is mining and logs a failure naming the card if not, with a unit test on the state machine (paused -> resumed -> every enabled card mining within one tick) and the known-failed case of PC 2's 21:25:11Z log; the defect left PC 2's miners off for 20 minutes tonight and the Mac's miner off after pause+resume this afternoon. If its commit is not in hand when the app tree is cut, it is first on 0.3.12's list and the status file says so. 0.3.11 carries program_class_v3 AND proving v1 together: one override object, one digest, one publish, the same gates for each half (its fast-time harness green on the final tree, suites on PC 2); both activation heights set at publish by the same rule (tip + 14,400, checked >= 10,800). If one half is not ready when the other is, the ready half ships as 0.3.11 and the other as 0.3.12; the status file says which.
Merge rule for the ship (C31): the app branch proving-v1 (e0de2ab) rewrote docs/evidence.md row 16 (WITHDRAWN, the 24 GB measurement) and the litepaper's proving sentences; ca2-coord carries the older row 16 and its own litepaper edits, so at the merge take proving-v1's row 16 and its proving sentences, and ca2-coord's everything else; the stale row must not win by accident. fud-close (the ledger closer's main-repo branch: 45 public-text fixes on the site and litepaper, spec 8.3 and 8.8, two CI checks, the relay fixes; a merge-tree onto ca2-coord shows 0 conflicts) is NOT in 0.3.11: the tree closed at 23bc2b2 (its workers, DMG and PC build job carry it) before the branch reached the ship order, and the shipper takes no late branch (the 0.3.10 rule); fud-close heads the next cut's list, with a coupling the next cut must respect: fud-close's worker change for ledger M28 (packfile.h's kernel_sha256 check, host.c refusing a pack that fails it) pairs with the fork-side miner change on ledger-fixes 3d4ec451 that stamps the hashes into program.json; the workers without that miner commit refuse every pack, the miner without the workers is harmless, so both go in one cut or the worker half of M28 is held back. fud-close tip 647b08c (its checks green); ledger-fixes is not yet rebased onto 89dfcb95 (two conflicting files: igneum/miner/src/main.rs, protocol/flows/src/ibd/proof.rs). The fork-side ledger-fixes branch (from release-0.3.6, not 21d4c73c) is NOT in 0.3.11: it rebases onto the 0.3.11 fork for the next cut. Branches merged into the 0.3.11 main tree beside ca2-v3 and ca2-coord (tooling, no consensus): consequences (the reviewer's rows), bash-body-check (7adb1ca, 6805125: tools/ci/bash-body-check.sh with fixtures and a self-test in ci.yml, the convention paragraph in packaging/README-ship.md, `publish-jobs.sh add --kind run` running the bash-body and prover-socket checks before signing; add/add with proving-v1 c2544be on tools/ci/prover-socket-check.sh: take bash-body-check's superset and proving-v1's ci.yml step; tools/amd-prove/pc1-cpu-prove.ps1 goes on the allow list, it starts no GPU server), amd-prove (f1d7a7d), card-lifetime (1fecfe2). Next-cut list (not consensus, not in 0.3.11 unless a one-file app change with tests): ota-k2 (branch ota-k2, commit c722579e, "OTA: the second signing key (K2) with revocation": manifest.rs, ota.rs, jobrun.rs, jobs.rs, inputs.rs, ota-sign.rs, engine.rs, the publish scripts, tools/keys, docs/security/keys.md section 4; ships signed with K1; K2's public half is empty until the project lead runs tools/keys/keygen-k2.sh), ember-tune 8ab9068 (the second-engine rule: no pipe into a second engine, its process tree killed at the end and on the budget, the installed app's miners restarted after; applied to ember-tune-pc1.ps1 and sweep-5090.ps1; tools/ci/second-engine-check.sh in ci.yml, shown to fire on a known-bad playbook and pass the fixed pair; every Cmd::Quit stamped with its source; the AMD gmax offset fix); rig-install (branch rig-install dd632c1, done) with two follow-ups that belong with 0.3.11 if the proving half ships then (a rig that cannot prove defeats the point), else 0.3.12: (i) the signed public manifest names no Linux package, so the installer verifies the HiveOS tarball through the unsigned downloads sidecar behind a flag; fix = `publish-public.sh --hive` adds a `platforms.linux` entry and re-signs (the apps ignore the extra key); (ii) the published Linux package carries no prover binaries and no key-hash / sign-record miner, so the rig's prover unit idles in "setup"; fix = a Linux prover build (sp1 host, pinned guests) in the cross-build set and the package; pool-v0 (its own service, no app change), repro-bench; per-day dataset reuse in the CUDA and OpenCL workers (a Day object split out of Pair: cache freed after the build, dataset kept across prepares of the same day; the Metal worker already does this; first item after the publish, for 0.3.12; until then the integrated tier mines v3 with a restart per epoch); the rig miners' `--exit-on-seed-change` path (exit 42, re-export on restart) replaced by prepare-ahead before any epoch shorter than an hour can be drawn (layer 9 precondition, consequences C20); fork-side pack-loop 05ef0fa3 is merged into the v3 node branch because v3 touches the same miner paths.
## 9. After the publish: Counter ASIC 3.0
The ASIC-history agent sends its ranked additions; they are measured the same way, folded into the class as v4 behind its own activation, same gates, same rollout; `docs/plans/counter-asic-3.md`.
## 10. Decisions table (superseded by section 6; kept for the options)
| Decision | Options | Recommendation | Source |
|---|---|---|---|
| The width rule (layers 1 and 2) | 4 B fixed; 16 B; 64 B; the per-program mix | `<the readwidth table's decision rule: the widest read that keeps every card latency-bound with margin on the 5090>` | `docs/plans/read-width.md` |
| The scratch share and size (layer 3) | 0, 12.5, 25, 50% RMW at 32 or 128 KB per warp | `<from the readwidth table and docs/analysis/scratch-soundness.md>` | the same |
| The hot table (layer 5) | 32, 64, 96 MB; replaced or added | DECIDED (5 October 2026, 21:36 UTC, delegated): OUT of v3. The rule was g at or above 0.97 on both PC cards in the added form; measured g is 0.87 / 0.85 / 0.84 on the 5090 and 0.84 / 0.81 / 0.80 on the 9070 XT at 32 / 64 / 96 MiB: neither card keeps even the 32 MiB table resident while the dataset streams, and the replaced form helps the on-die-cache chip. Layer 5 stays a measured option for 3.0 | `docs/plans/hot-table.md` (ca2-cache): 5090 MH/s v2 136.1; replaced hot32k4 146.6, hot64k4 140.8, hot96k4 138.5, hot64k2 137.5, hot64k8 163.6; added hot32k4a 118.7, hot64k4a 115.4, hot96k4a 114.4. 9070 XT v2 18.15; replaced 19.79 / 18.73 / 18.33 / 18.17 / 22.32; added 15.27 / 14.62 / 14.56. M5 Max added 0.93 / 0.87 / 0.83. All eight packs bit-exact on the 5090 and the 9070 XT with the Mac's fingerprints (21:29 to 21:35 UTC, no restart straddled) |
| The era draws (layers 4 and 8) | in, if the min-to-max spread across six drawn eras is under 5% per card; item size fixed at 4 B, draws of stride, interleave and the working-set window at or above 256 MiB | DECIDED (5 October 2026, 21:58 UTC, delegated): IN. Spread over six eras: RTX 5090 1.3% (136.18 / 136.44 / 138.01 MH/s against v2 137.2), RX 9070 XT 3.2% (18.61 / 18.93 / 19.21 against 18.09), M5 Max 0.8% (28.35 / 28.48 / 28.58 against 27.68); all under 5%; latency-bound share 1.01 / 0.95 / 1.06; every pack's 2^24 fingerprint equal on all three vendors; the CPU verifier 1.00 to 1.02 of v2 within one binary | `docs/plans/era-layout.md` (ca2-era 78c0ee4; PC 1 jobs fetch-ca2-era-20261005 and run-ca2-era-pc1-20261005, 301 s, both cards restored). Chip line: 512 B read per hash, 120 to 128 distinct lines, the mirror is the whole dataset every hour; the interleave's value against a chip with a programmable address decoder is nil (stated), the stride is a bijection with no cryptanalysis yet |
| The cache schedule (layer 6) | flat 256 MiB or a growth schedule | DECIDED (5 October 2026, delegated under "execute the full 2.0 plan, tonight 1-8"; the project lead confirms for the public testnet genesis; the mirror's mm^2 are being corrected to shipped-product density, 2 to 3x the bit-cell figures, conclusion unchanged): option C, the cache doubles when the dataset doubles: 256 MiB at genesis, 512 MiB at year 4, 1 GiB at year 12. Verifier fill on one M5 Max core at 0.2 s per 256 MiB (spec 1.12): 0.2 s, 0.4 s, 0.8 s at each step, under 1 s at every step of the schedule; verifier memory 256 MiB, 512 MiB, 1 GiB | `docs/analysis/sram-mirror.md`: a 256 MiB mirror is 54 mm^2 at N2 (0.0175 um^2 cell, array factor 0.70), about $26 per good die, approximate |
| Layer 7 | reserved family, unlock by era height or 90% signal | DECIDED (5 October 2026, delegated): reserve family R1 = mm8 (uint8 8x16 by 16x8 tile per unit, unsigned bytes), W_new 4, unlock at era 4 or 90% signal, the emulation rule in spec 1.13.2; switched off, no consensus effect tonight; the 5090 and 9070 XT dp4a numbers when the PCs free (PC 2 first) | `docs/analysis/int8-matrix-family.md`: native on PTX mma.sync, AMD WMMA iu8, Metal 4 matmul2d; dot4 emulation on Apple 1.6x (unsigned) |
| The activation height N4 | the rule of section 3 | N4 = N5 = 154,800 (the shipper, 22:45 UTC: DAA 136,967 at 0.965 blocks/s puts the publish near 140,200, tip + 14,400 near 154,600, the first multiple of 3,600 at or above is 154,800; one number for both switches); the 10,800 floor holds until DAA 144,000, about 00:45 UTC, re-pinned and rebuilt past that | release-0.3.11 23bc2b2 (packaging/mac/packaged-config.sh, the nine-field line) |

View file

@ -0,0 +1,633 @@
# Counter ASIC 2.0: status
Coordinator's running status for the plan in `docs/plans/counter-asic-2.md`. Rewritten every 45 minutes while the work runs. Times UTC, 5 October 2026 (night). Heading times before 20:40 were corrected at 20:42 from the commit clock (the coordinator had written them from a guessed clock, up to 2 h 40 min ahead; corrected again at 20:49 for the 20:42 to 20:47 entries); every entry's true time is its commit's author time in UTC, and from 20:49 every heading is stamped from `date -u`. Base for every ca2 branch: `readwidth` at 019b014 (the LoadClass flag, fold_words, the scratch op, the three emitters, 20 packs).
## 19:55 first status (the 19:50 start was cut off by exhausted credits at about 19:58 before any sub-agent work landed; respawned at 19:55 on the restart)
| Layer | Branch | Agent | State | Numbers so far | Blockers |
|---|---|---|---|---|---|
| 1, 2, 3 (widths, mix, scratch) | readwidth 019b014 | a451c9935bfb1bc19 (not ours) | Mac Metal rows in; Apple OpenCL next, then one job per PC; scratch re-run at 32 and 128 KB per warp | CPU verifier, M5 Max one core, avg of 50, ms per warp: v2 0.604, w16 0.610, w64 0.630, w64x4 0.160, mixA 0.620, mixB 0.614, scr0 0.600, scr2 0.534, scr4 0.457, scr8 0.314. Metal M5 Max hash rate (5 x 2^24, GPU time, 3 of 3 vectors on every pack): v2 27.7 MH/s, w16 28.3, w64 28.2, w64x4 109.7 (32 loads of 64 B), mix 50/35/15 over 6 programs 25.4 to 28.4 (median 26.7), mix 25/50/25 over 6 programs 22.0 to 25.2 (median 24.4). Scratch at 1 MiB per warp, 1,024 warps: 23.1 / 18.2 / 16.3 / 14.9 MH/s at 0 / 12.5 / 25 / 50% RMW, superseded by the cap | holds the Mac measure lock and both PCs first |
| 4 + 8 (era layout, working set) | ca2-era | a452664c512c73b9b | design, respawned 19:56 | none | PC time behind readwidth |
| 5 (cache-sized second table) | ca2-cache | a5271cf269757b118 | design, respawned 19:57 | none | PC time behind readwidth |
| 6 (SRAM schedule) | ca2-analysis | a5c6bc2dfcc4613ef (respawned 19:58) | cited analysis | none | none |
| 7 (integer matrix family) | ca2-analysis | a5c6bc2dfcc4613ef | design; dp4a throughput owed unless PC time frees | none | Apple int8 path to check against Metal docs |
| 3 soundness | ca2-soundness | a548aadeefd1ab3b2 (spawned 19:59; sizes 32 and 128 KB per warp) | analysis + tests | none | none |
| Integration v3 | ca2-v3 | after the above | waiting | none | the readwidth table and the four branches |
Budget rule received from the coordinator at the restart: the per-warp scratch is capped so the whole working set (1 GiB table + hot table + scratch for every resident warp + buffers) stays under 6 GB on an 8 GB card, which puts the scratch in the tens of KB per warp; the layer 5 hot table shares that budget.
Decisions for the project lead so far: none. Nothing here touches consensus or any live node.
## 20:05 readwidth PC jobs out; the scratch finding
Base moved: every ca2 branch rebases onto readwidth b970dda (scratch per warp is a class parameter, 32 or 128 KiB; the SCRATCH_* constants are gone; Metal pack harness `proto-metal/packbench.swift`; OpenCL `--bench-pack`). The coordinator branch is rebased; the four agents were told.
Readwidth PC jobs published 20:02:51Z: `fetch-readwidth-20261005` (both PCs), `run-readwidth-5090-20261005` on PC 2 (6 to 10 min once it starts; a build job from another session is queued ahead of it), `run-readwidth-9070-20261005` on PC 1 (10 to 15 min; only the gfx1201 card is switched off). My layer 4, 5 and 7 PC jobs queue behind these.
Finding that bears on layer 3 (readwidth, Metal, M5 Max, capped sizes): the scratch read-modify-writes are cheaper than the dataset loads they replace, so the rate RISES with the RMW share.
| Class | MH/s |
|---|---|
| v2 | 27.7 |
| scr0k32 (control) | 28.1 |
| 32 KB per warp, 12.5 / 25 / 50% RMW | 25.4-26.1 / 29.4-31.7 / 44.4-49.1 |
| 128 KB per warp, 12.5 / 25 / 50% RMW | 24.3-26.2 / 26.4-28.1 / 34.1-35.4 |
Reading (readwidth agent): 4,096 warps x 32 KB = 128 MB sits in the chip's caches. Consequence for the decision: a scratch that fits a GPU's cache fits a chip's SRAM at the same size, so at the capped size the writes cost everyone a cache-bound op in place of a latency-bound load. Passed to the soundness agent: what size would make the writes cost DRAM latency, whether that fits the 6 GB cap, and whether the RMW share should be added to the 16 dataset loads rather than taken from them.
## 20:08 mandate: v3 on the devnet tonight, by the project lead's rules
the project lead has gone to bed and delegated the three decisions for the DEVNET only (not the public testnet): width, per-load mix and scratch share by the rules now written in `docs/plans/counter-asic-2-rollout.md` section 6, the activation height = devnet tip + 14,400 at publish (checked >= 10,800), published the way finality v3 was. Six gates before any publish (rollout section 7): bit-exact v3 on the three cards; the CPU verifier exact on 1,000 random GPU hashes per card; the soundness suite green with the new scratch tests; the fast-time 3-node network mining across a v3 activation with 0 rejected blocks and 0 forks; Windows and Mac workers from one commit; the node change on a fork from the 0.3.10 tip (21d4c73c) with suites green on PC 2. Release 0.3.11 through the shipper's pipeline; the 0.3.10 shipper (ae892a8b0f78fe31c) has been asked for its state and the handoff. If a gate fails: stop, write why here, do not publish.
Node build note for the integration: the node links `igneum-pow` by path (`../../../../igneum-pow` from `consensus/pow` and `igneum/miner`), so the node fork worktree for v3 must live under the v3 worktree's `vendor/` so that the path resolves to the v3 crate, not master's.
After the publish: Counter ASIC 3.0 from the ASIC-history agent's ranked additions (a202a09dcd24ba1d3), as class v4 behind its own activation, same gates, `docs/plans/counter-asic-3.md`.
## 20:08 the shipper's answer, the node fork convention
0.3.10 (shipper ae892a8b0f78fe31c): staged and blocked on GitHub's Actions incident (run 37365130137 queued since 19:42:43Z under a re-dispatching watcher); nothing on the network has moved, the live manifest is still 0.3.9. Once CI is green: ship (5 min), update-now (apps restart 1 to 10 min later), hand nodes and seed (5 min), digest sweep; 0.3.10 finished about 30 min after green. The app version per machine on the console's cards is the restart signal for re-running any straddling measurement. HiveOS is published by the ship's --public step, nothing separate.
Node fork for v3: base on COMMIT 21d4c73c (release-0.3.10 in vendor/igneum-node = release-0.3.6 a24ab01a + housekeeping 4fb32865 + tx-gossip e242acd0 + c4-fix e18f1e0e). Fork-side pack-loop 05ef0fa3 (the miner's pack check, exit 44) is not in it and touches the same miner paths as v3 (export-pack, the job line): merge it. Because the node links igneum-pow by relative path, the v3 node worktree goes under the v3 worktree: `git -C /Users/joshm/Projects/igneum/vendor/igneum-node worktree add /Users/joshm/Projects/igneum-wt-ca2-v3/vendor/igneum-node-ca2 -b ca2-v3-node 21d4c73c`.
0.3.11 shipper: the coordinator assigns it (the 0.3.10 shipper stops at its report). Inputs the ship needs: the fork commit with its PC 2 suite results recorded, the main tip, the override object with every switch (the four live fields plus program_class_v3_activation_daa), the expected digest read on a 22-s scratch node, the deadline note ("program class v3"), the activation height, the one-line changelog, and whether the pinned proving guest changes (it does not: the prover has no igneum-pow dependency; confirmed by grep of proving/igneum-prove Cargo files).
## 20:10 readwidth round 2 on the PCs; the width arithmetic under the project lead's rule
Round 1 of the readwidth PC jobs refused every pack (the workers demand a 32-byte chain seed; the experiment packs carried string seeds); fixed at readwidth 1ea7a52 (packfile.h), republished 20:09:19Z as `run-readwidth-5090-20261005c` (PC 2, about 6 min) and `run-readwidth-9070-20261005c` (PC 1, about 10 min). The era and cache agents were told to rebase onto 1ea7a52 and to prove their packs load on the Mac OpenCL host before any PC job. Every ca2 branch now bases on 1ea7a52.
Probe ceilings from round 1 (dependent reads per second at 1024 MiB, device time):
| Card | 4 B | 16 B | 64 B | Stream |
|---|---|---|---|---|
| RTX 5090 | 17.5 G | 18.0 G | 9.1 G (584 GB/s) | 1,579 GB/s |
| RX 9070 XT | 2.42 G | 2.43 G | 2.47 G (158 GB/s) | 636 GB/s |
the project lead's width rule applied to the ceilings alone (the measured v3 rates will replace this when the table lands): a 128-load hash at 64 B reads 8,192 B; at the 5090's 64 B ceiling that is 71 MH/s and 584 GB/s, 37% of its stream bandwidth, over the one-third margin the rule sets; at 16 B it is 2,048 B per hash, 141 MH/s and 288 GB/s, 18%, inside the margin; 4 B is 9%. On the 9070 XT every width costs the same line fetch (2.4 G/s), so 16 B is where AMD gains 4x the bytes per hash at no cost and the 5090 stays latency-bound with margin. Provisional width under the rule: 16 B (w16), pending the measured rates and the latency-bound share per card.
Added deliverable (20:11): the public description in four levels, `docs/plans/counter-asic-2-public.md` (aec53bb): levels 1 and 2 are written as copy; level 3 carries the bench table with `[owed]` markers for every number not yet measured; level 4 lists the documents. The integration branch applies levels 1 to 3 to `site/index.html`, `site/litepaper.html` and `site/bench.html` with the final numbers.
## 20:15 layers 6 and 7 landed; the node-fork agent started; 0.3.11 scope
Branch ca2-analysis (5d5ba15, f59708d). Layer 6: no cache growth rule exists in the spec; cited bit cells N7 0.027, N5 / N3E / Intel 18A 0.021, N3B 0.0199, N2 0.0175 um^2, array factor 0.70; the 256 MiB mirror is 83 / 64 / 54 mm^2 at N7 / N5 / N2 (74 at N2 with a 96 MB hot table), $13 to $26 of silicon per good die (approximate). The mirror was never unaffordable; the cache's job is to stay above GPU L2 (5090 96 MB, GB202 128 MB). Recommendation C for the project lead (gate 1): cache doubles when the dataset doubles. Layer 7: Metal 4 matmul2d has int8 x int8 -> int32, so a unit-level mm8 tile is native on all three vendors; per-lane dot4 is emulation on Apple (M5 Max: 548 G unsigned dot4/s against an 880 G ALU chain, 1.6x; signed 4.7x). Reserve R1 = mm8, W_new 4, unlock era 4 or 90% signal. Owed: the 5090 and 9070 XT dot4 probe (job prepared: relay/playbooks/dot4-probe.ps1, exe sha256 5adaeb1a...6416f4; publishes when a PC frees).
Node-fork agent a3f505a9d981300cd started 20:50: ca2-v3 (igneum-pow seam: ProgramClass, Epoch::from_chain_seeds, generator 3 in the program id, class in the pack) and ca2-v3-node (vendor/igneum-node-ca2 under the ca2-v3 worktree, from 21d4c73c, with pack-loop 05ef0fa3 merged): the field in Params, OverrideParams and the digest, the epoch-boundary rounding, the era stand-in, the job line, the fast-time gate script.
0.3.11 scope (coordinator, 20:52): carries program_class_v3 and proving v1 together (one override object, one digest, one publish; each half under the same gates); the ready half ships as 0.3.11 and the other as 0.3.12 if one lags. Next-cut list recorded in the rollout plan section 8a.
## 20:16 decisions recorded: layer 6 option C, layer 7 R1 = mm8; the chip headline
Layer 6 DECIDED (delegated; the project lead confirms for the public testnet genesis): option C, the cache doubles when the dataset doubles (256 MiB genesis, 512 MiB year 4, 1 GiB year 12); one-core fill 0.2 / 0.4 / 0.8 s, under 1 s at every step. Layer 7 DECIDED: reserve R1 = mm8, unsigned, W_new 4, unlock era 4 or 90% signal. The chip model's headline now names the on-die-cache recompute chip (54 to 83 mm^2) as a row per variant; the scratch share is chosen as the smallest share at which that chip's gain falls under 1.5x, else said plainly and the public "under 2x" claim qualified. The soundness agent carries that table (its question 2) with M16's mixer multiplier beside it.
## 20:18 correction to the SRAM mirror figures (chip-economics research, cluster D)
Bit cell x 0.70 understates real die area. Shipped cache dies: AMD 3D V-Cache 64 MB on 41 mm^2 at 7 nm (1.56 MB/mm^2, Tom's Hardware, Hot Chips August 2021); Graphcore GC200 900 MB on 823 mm^2 with compute (1.09 MB/mm^2); Groq TSP 220 MB on 725 mm^2 at 14 nm (0.30 MB/mm^2); TSMC N5 HD SRAM macro 31.8 Mib/mm^2 after about 30% assist overhead (SemiAnalysis, December 2022). A 256 MiB mirror is about 165 mm^2 at 7 nm on the densest shipped cache-only die and about 130 mm^2 at N5/N3E, not 54 to 83 mm^2; cost per die 2 to 3x the earlier figure; the conclusion (affordable for a funded chip) stands. The analysis agent is redoing the table with both columns; the soundness agent carries the corrected density into the chip row. Latency citations behind the latency-bound rule, to be added: DRAM row cycle 40 to 48 ns across DDR4, GDDR5, HBM2 (Li, Reddy, Jacob, MEMSYS 2018); latency 1.3x in two decades against bandwidth 20x (Chang 2017); no shipped mining chip used HBM or stacked memory.
## 20:19 proving v1 state for 0.3.11; PC 2 occupancy
Proving v1 (acd4f36bc2c07a4e2): fork proving-v1 b177718e on a24ab01a (told to rebase onto commit 21d4c73c now), app proving-v1 79bc820 on a93199a. Override fields proving_v1_activation_daa (tip + 14,400 at publish), proving_v1_segment_blocks 4, proving_v1_unproven_daa 600, proving_v1_aggregator_share_bps 1000; they enter the digest only once the activation is set. Harness: `tools/proving-v1/net.mjs --secs 1500` PASSED (21 checks) in 197.3 s on b177718e; rerun owed on the final tree. The pinned guests do not change. Shared files with ca2-v3-node: params.rs, daemon.rs, igneum/miner/src/main.rs, override-60x.json; both agents keep separable hunks.
PC 2 is held by the proving agent's memsweep-pc2-pv1 (about 20 min, miners stopped) and a second run (about 10 min). Queue after it: the ca2 node suites, then the readwidth, era, cache and dot4 measurement jobs. PC 1 is held by readwidth's run-readwidth-9070-20261005c until it reports.
## 20:21 sram-mirror.md revision 2 (ca2-analysis)
Two columns, headline = shipped-product density (AMD V-Cache 41 mm^2 per 64 MiB at N7, scaled by the bit-cell ratio), lower bound = bit cell x 0.70. mm^2 and $ per good die (D0 0.1 per cm^2, wafer prices approximate), headline / lower bound:
| Node, wafer $ | 256 MiB | 256 + 96 MB hot table | 1 GiB |
|---|---|---|---|
| N7, $9,500 | 164 / 83 mm^2, $30 / $13 | 226 / 114, $44 / $19 | 656 / 331, $224 / $75 |
| N5, N3E, 18A, $20,000 | 128 / 64, $46 / $21 | 175 / 89, $68 / $30 | 510 / 258, $306 / $111 |
| N3B, $20,000 | 121 / 61, $43 / $20 | 166 / 84, $63 / $28 | 483 / 244, $280 / $103 |
| N2, $30,000 | 106 / 54, $56 / $26 | 146 / 74, $81 / $37 | 425 / 215, $343 / $131 |
Year 10 at the 6% per year trend: 59 mm^2 for the flat cache (82 with the hot table), 8% of a 750 mm^2 die. One reticle holds 1.3 GiB (N7) to 1.9 GiB (N2); mirror share of a 750 mm^2 die at year 0: 14% (22% with the hot table), inside M16's 13 to 40% band. Recommendation unchanged: C. Latency section added (MEMSYS 2018, Chang 2017, the mining-chip memory-type note), marked as research the agent did not re-read tonight apart from the V-Cache figure.
## 20:23 layer 3 soundness landed: the scratch does not move the chip; scratch share decided 0
ca2-soundness (0d8f745 tests and trace hook, a465881 doc and bench-log). The on-die-cache recompute chip (N5 headline 128 mm^2, $46) at 50 T op/s: 333 MH/s against the 5090's measured 139.7, 2.4x; at 12.5 / 25 / 50% RMW replaced, chip 381 / 443 / 661 against 5090 projected 160 / 186 / 279, 2.4x each; added, 2.4x or more; 32 or 128 KB alike. The chip keeps the scratch implicitly in 80 to 320 B per lane (the verifier resets it per unit), needs about 530 units in flight, dense scratch 6.2 / 3.1 mm^2 at N5. Under the project lead's rule the scratch share is 0: layer 3 is NOT adopted into v3; the public "under 2x" claim is qualified (public copy level 3 rewritten). The measured lever is M16's mixer multiplier (x2 1.2x at 0.8 to 2.4 ms verify; x4 0.6x at 1.6 to 4.8 ms; 3.6x and 1.8x with a 3x fixed-function factor); whether x4 enters v3 tonight is asked of the coordinator; default: Counter ASIC 3.0.
Soundness results (Metal, M5 Max): 28/28 edge launches, 200/200 fuzz packs (91 s), 56/56 hand-model edge checks, 42/42 kernels pass the static scratch-mask check with 6 deliberate breaks caught, broken tag and broken lazy fill caught, fingerprint 8c07620f4d9adefd warp-count-independent; re-hit rates 2 to 33% above the birthday bound (slot addresses are register low bits); written words unbiased (worst 3.63 of 6 sigma). Verifier exactness needs a host contract (zero the arena at allocation and at the 32-bit tag wrap, tags from 1), which neither host gives today. Pre-existing on readwidth b970dda: verify::tests::fold_and_wide_fetch overflows under the test profile (wrapping_mul fixes it); passed to the readwidth agent with the class sweep.
Gate G3 note: the scratch tests (igneum-pow/tests/scratch.rs) join the v3 suite even though the class carries no scratch, parametric over the class; they guard the v2 path's scratch-free invariant at zero cost.
## 20:24 decided: M16 mixer x4 into v3; agent ca2-mixer started
Coordinator's decision under the project lead's delegation (recorded in the rollout plan section 6a): the mixer multiplier x4 and the cache growth rule (option C) enter class v3 behind the same activation; layer 3 stays out at scratch share 0, its soundness document and pack-contract tests kept. Agent af345b1e2c541ffbb (branch ca2-mixer) implements `mixer_mult` as a class parameter (m mixer applications per round, the 8 dependent reads unchanged), the `cache_log2_words(day)` schedule (doublings at years 4 and 12 with the dataset stepping to the next power of two), re-cuts the v3 dataset vectors, re-runs the soundness suite, measures the verifier (v2 0.604 ms per warp; v3 expected 1.6 to 4.8 ms) and the 1 GiB build on the Mac, prepares the 5090 and 9070 XT build-time job, and writes docs/analysis/chip-model-v3.md with the combined headline row (fixed-function factor included). The claim on the site reads "under 2x" only if that row does; else qualified, with the mixer x8 and the hot table named as the next levers.
Agents now: ca2-era (a452664c512c73b9b), ca2-cache (a5271cf269757b118), ca2-node (a3f505a9d981300cd), ca2-mixer (af345b1e2c541ffbb). Done: ca2-analysis, ca2-soundness. Waiting: the readwidth PC table; PC 2 (proving memsweep runs) and PC 1 (readwidth 9070 round).
## 20:27 the readwidth table landed; layers 1, 2, 3 decided; layer 5 measured on the Mac and redesigned
Readwidth e752fc7 (`docs/plans/read-width.md`), bit-exact on Metal, Apple OpenCL, the 5090 (NVRTC) and the 9070 XT, both PCs released. MH/s (latency-bound share):
| Class | RTX 5090 | RX 9070 XT | M5 Max | Gap |
|---|---|---|---|---|
| v2 (128 x 4 B) | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x |
| w16 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x |
| w64 | 71.9 (0.58, 37% of stream) | 17.59 (0.78) | 28.27 (1.03) | 4.1x |
| w64x4 (32 loads) | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x |
| mix 50/35/15, six programs, spread of median | 18.8% | 7.4% | 11.3% | |
| mix 25/50/25 | 22.3% | 5.5% | 8.1% | |
| scratch 32 KB at 12.5 / 25 / 50% | -18 / -21 / -12% | -18 / -22 / -21% | -7 / +12 / +74% | |
| scratch 128 KB | -21 / -30 / -48% | -21 / -27 / -33% | -7 / -1 / +25% | |
Decisions (the project lead's rules, delegated): layer 1 keep v2 (w16 passes the rule but closes nothing and does not move the chip row; the vector re-cut is not worth it); layer 2 out (spread over 5% on every card); layer 3 out (scratch share 0). The AMD gap is the card's dependent-read rate (2.4 G/s at every width), stated for the user tiers in the rollout plan 6b.
Layer 5 (ca2-cache 53ef59f, 011cc0a, 86726cd, 65bc7a7; `docs/plans/hot-table.md`): five packs bit-exact on Metal and Apple OpenCL (96/96 each). M5 Max rates against v2 27.68: hot32k4 x1.22, hot64k4 x1.12, hot96k4 x1.05, hot64k2 x1.00, hot64k8 x1.71; verifier 0.344 to 0.560 ms against 0.626; hot fill per epoch 24 / 46 / 73 ms on one core, 0.07 / 0.15 / 0.22 ms on the GPU; Apple OpenCL probe 32 / 64 / 96 / 1024 MiB 21.7 / 12.8 / 12.3 / 3.50 G loads/s. Redesign ordered: hot loads ADDED beside the 16 dataset loads (the replaced form lets the on-die-cache chip skip item derivations and worsens the gain); the agent re-measures the added form and rebuilds the PC job. PC 1 is given to the dot4 probe (under 15 min), then to the era agent, then the hot-table job; PC 2 stays the proving agent's.
## 20:27 layer 9 added; the era draw passes Mac bit-exactness; two class bugs in the harnesses; C1
Layer 9 (the project lead: faster program changes): the epoch length becomes an era parameter in the genesis reserve, 1 hour at launch, 10 minutes to 2 hours by draw or 90% signal, reserve-only tonight; an agent (ca2-epoch) designs it beside layers 4 and 8 and measures the compile-ahead cost per card at a 10-minute epoch, the seed-path consequence and the FPGA threat it answers; one row in the rollout plan section 6, one in the level 3 numbers. Spawns when the dot4 probe frees its slot.
ca2-era mid-way (a452664c512c73b9b): era draw behind LoadClass::era (EraParams beside mix, load_slots, scratch, scratch_kb; verify::load_index; memhard::Layout; one load form in the three emitters; --era, --era-widths), 54 crate tests green, pinned packs byte-identical; six era packs bit-exact on Metal (packbench 6/6), Apple OpenCL (6/6, same fingerprints) and the CUDA emulation (6/6); re-exporting with the width pinned at 4 B and rebasing onto e752fc7; PC job in about 20 minutes. Two class bugs found and fixed on its branch, both outside its layer: (1) proto-cuda/nvrtc/packfile.h re-derived the seed words as attempt 0 only, so ANY pack with IGNEUM_PROGRAM_ATTEMPT >= 1 (5.14% of chain epochs under v2) is refused by the one-click workers with "the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT", the error PC 1 logged on 5 October and attributed to the export race; fixed attempt-aware, with a tampered-attempt refusal test. This fix must ship in the 0.3.11 workers whatever else does. (2) proto-cuda/host.cu and proto-opencl/host.c derived dataset words on the host as mh_item(w >> 4)[w AND 15] instead of the pack's mh_word; fixed.
Consequences review (a20f8c09b90016cc7) C1, C10, C11: C1 recorded in the rollout plan 7a (0.3.11 on every prover before 16:00Z on 6 October, else the fee switch is republished at tip + 86,400); C10 was resolved by the mixer x4 decision (20:23, 20:24 entries), the pool-core and node verification numbers are being measured on ca2-mixer; C11: AMD RDNA 4 is about a seventh of a 5090 on this hash by dependent-read rate, 4.9x the electricity per hash and 2.2x worse per pound at list prices (0.032 against 0.072 MH/s per pound, read-width.md 4.1, approximate); the card's memory system, not a tuning gap; MH/W and MH per pound columns (list prices, approximate) go into the final table and the level 3 page; the line-width question is a 3.0 question since no width closes the gap without making the 5090 bandwidth-bound.
## 20:29 C1 decided for the morning; one packfile.h fix for 0.3.11
C1 (the fee switch H = 210,000 at about 18:45Z on 6 October by the 22:32:54Z DAA read, 1.002 DAA/s since 15:40Z; a 0.3.10 prover's shards are vetoed from H because its export RPC carries no daaScore and no fee schedule). Decision: (a) 0.3.11 carries the proving v1 fork eb32c645 (on 21d4c73c; protocol 15, message 75) and the app branch b399708 (on 5b0d54f) with H unchanged at 210,000, published as the fleet sweep. THE 16:00Z CHECK on 6 October, for whoever holds the morning: open the console's machine cards; if every prover (PC 1 ae432dc7, PC 2 1ccfe586, the Mac) shows 0.3.11, H stands and nothing is done; if any prover is not on 0.3.11, the publisher republishes H = tip + 86,400 by the fee-switch plan's rule (one digest flip, every node in one sweep: manifest, update-now, hand nodes, seed) before about 18:45Z. The proving agent recommends (b) unless (a) is certain; the check decides it.
The attempt-0 packfile.h bug (the era agent's find) is the one that took the fleet down at 18:23Z (epoch 34, attempt 1); it is already fixed attempt-aware on pack-loop af983a7 and merged into the 0.3.10 tree. Rule for 0.3.11: one derivation, one test set: the era and cache agents build their workers on that packfile.h and keep only tests that add a case; the host.cu / host.c mh_word derivation fix (new, the era agent's) stays with its own test.
## 20:31 card lifetime merged; cache freed after the build; growth mapping (b) recommended
`docs/analysis/card-lifetime-2026-10-05.md` (branch card-lifetime 1fecfe2) merged into ca2-coord. Decided (delegated): the GPU frees the 256 / 512 / 1,024 MiB cache after the daily dataset build (the hash never reads it; the rebuild costs 0.67 ms fill + 13.4 ms build on the 5090 at x1, about 54 ms at x4, owed); hot-table.md's resident reading is corrected, era-layout.md's freed reading stands. Recommended for the project lead: growth mapping (b), power-of-two steps at years 4, 12, 28, 60 with AND MASK, the only mapping under which the litepaper's "4 GB about four years, 8 GB more than a decade" holds (4 GB: year 4 under (b), 1.0 to 1.5 years under (a); 8 GB: year 12 or 6.3 to 7.5; 12 GB: year 28 with the cache freed; 24 GB: year 60). Public lines and the evidence row go on the integration branch (rollout plan 6c); hot-table.md line 73 (the 8 GB row counted a 5090's warps) goes to the cache agent.
## 20:32 layer 9 agent started; the dot4 probe is on PC 1
ca2-epoch (a32a3ece66c02417a): the epoch length as an era parameter (600 to 7,200 DAA s, base 3,600; draw or 90% signal; the VDF rule; the difficulty-window constraint; the FPGA threat with citations), Mac compile-ahead measured now, the 5090 and 9070 XT compile times cited from the bench log, docs/plans/epoch-length.md. The dot4 probe job is running on PC 1 (ca2-analysis tip ee42d7c carries the playbook; the 5090 confirmation is the agent's watch). PC 1 queue after it: the era six-pack job, then the hot-table added-form job. PC 2: the proving agent's, then the ca2 node suites.
Agents running: ca2-era, ca2-cache, ca2-node, ca2-mixer, ca2-epoch; ca2-analysis watching its PC job. Done: ca2-soundness.
## 20:38 layer 7 complete on all three cards (ca2-analysis ee42d7c); PC 1 free
dot4 probe on PC 1 (jobs fetch-dot4-20261005, run-dot4-20261005, exit 0, 101 s; both cards restored and mining; 5090 SM clock 2,505 MHz before and after):
| Device | ALU chain, G steps/s | signed dot4 emulation, G dot4/s | dot4 instruction, G dot4/s | emulation vs instruction |
|---|---|---|---|---|
| RTX 5090 (NVIDIA OpenCL 3.0, driver 617.14) | 8,753.5 | 1,239.1 (7.1x an ALU step) | 7,453.6 (inline PTX dp4a.s32.s32, 1.17x) | 6.0x |
| RX 9070 XT gfx1201 (AMD-APP 3683.0) | 701.4 | 480.8 (1.46x) | 664.3 (__builtin_amdgcn_sudot4, 1.06x) | 1.38x |
| gfx1036 (RDNA 2 iGPU) | 40.6 | 15.8 (2.6x) | sudot4 does not build (needs dot8-insts) | |
| M5 Max (Metal) | 879.8 | 188.2 (4.7x); unsigned 548.2 (1.6x) | none exists | |
All bit-exact against the CPU reference. One dp4a costs about one ALU step on NVIDIA and AMD; the 5090 is 12.5x the 9070 XT on the ALU chain and 11.2x on dot4 (the family does not widen the AMD gap), 10x the M5 Max on the chain and 13.6x on dot4 (Apple's emulation widens its gap 1.4x). At W_new 4 the family adds about 21 ops per hash per lane; hash-rate losses expected under 5% on every card (to be measured with the family live). Owed: the CUDA __dp4a cross-check (needs nvcc), Metal 4 matmul2d int8 on the M5, sdot4 on RDNA 2. cl_khr_integer_dot_product is listed by no driver we own.
PC 1 is free: the era six-pack job goes next when its package arrives, then the hot-table added-form job.
## 20:39 PC 1 scheduler (the coordinator's role from now): the queue
Rule: one PC 1 job at a time; an agent asks by message before publishing, gets a "go PC 1" from this coordinator, and reports when its RESULT lines are in and both cards are restored; a hash-rate or power number taken while another job holds a card is not a number. The CPU-only job runs only in a slot where no measurement overlaps it.
| # | Job | Agent | Cards | Length | State |
|---|---|---|---|---|---|
| 1 | Era six-pack (layers 4 and 8), 5090 and 9070 XT | ca2-era a452664c512c73b9b | one card at a time | about 15 to 20 min | waiting for the package (packs on the pack-loop packfile.h) |
| 2 | Hot table, added form, probe 32/64/96 MiB plus packs | ca2-cache a5271cf269757b118 | one card at a time | about 15 min | waiting for the re-measured Mac rows and the rebuilt zip |
| 3 | Reproducible benchmark run | a0b9f574775ef1693 | one card at a time, 120 s per card | about 5 min | queued |
| 4 | AMD sweep on the 9070 XT (core clock and power steps) | a01dcb34ae16d867c | 9070 XT only; the 5090 keeps mining | about 20 min | queued |
| 5 | Ember Tune end to end, both cards | a855dcc4bd05e0615 | both | to be stated | queued after the AMD sweep |
| 6 | AMD-proving CPU fallback (CPU-only SP1 run, both cards mining) | a39db54d4de4af51e | none; loads the CPU | to be stated | last, or in a gap where no measurement runs for its whole length |
If job 1's package is more than 15 minutes away when job 2 is ready, job 2 goes first; the short job 3 fills any gap of under 10 minutes between packages.
20:39. No more agents are spawned tonight (the project lead: no unnecessary credits); the running ones finish. PC 1 queue change: the AMD-proving CPU fallback's small fixture (block-56-transfers-3shards, minutes) runs NOW in the gap before the era package; its S_p shard (block-338-shard1, up to 30 min of every core) stays job 6, last. Next-cut list gains the rig installer's two follow-ups (the Linux manifest entry, the Linux prover build), tied to whichever release carries the proving half (rollout plan 8a).
## 20:42 the node switch is written; the Metal worker is a gate item
ca2-node (a3f505a9d981300cd): ca2-v3 commits d2cd6e1 (pack-loop af983a7 merged: packcheck.rs and the attempt rule; one packfile.h conflict resolved), 50d5c86 (the seam: ProgramClass { V2, V3 }, V3_CLASS placeholder w16, generator 3 in the program id, Epoch::from_chain_seeds, Epoch::chain_program, IGNEUM_PROGRAM_CLASS and IGNEUM_ERA_SEED_HEX in the packs, packcheck refuses wrong class / era / generator; v2 packs byte-identical; 53 tests), 9ed787e (workers: packfile.h reads class and era, CUDA and OpenCL workers take `class=v3 era=<hex>` tokens on job and prepare lines, pair identity includes them, mismatch answers `need`; 13 packfile checks). Fast-time gate script infra/fast-time/class-v3.mjs written, not run (needs the fork binaries). Node (ca2-v3-node, uncommitted until cargo check passes, queued behind the measure lock): the field in Params, OverrideParams, override_params and the digest (unconditional, its own statement; 11-entry digest test), the daemon line, program_class_for_epoch_at (v3 iff 3600 e >= N4, first epoch = ceil), POW_ERA_BLOCKS 15,552,000 and POW_ERA_LEAD 7,200 with the era stand-in (era 0 = genesis), EpochSeeds { epoch, day, class, era }, template and RPC fields 12 to 16, the miner's job line and seeds.txt. PC 2 command ready (run from the ca2-v3 worktree so push-build-inputs.sh packs the v3 igneum-pow); held until "ready for PC 2".
Gate item found: main.swift refuses v3 lines "until Swift has generator 3"; the Mac worker never regenerates a program (igneum-miner export-pack writes the pack), so the fix ordered is to accept the pack's class and era against the line, as the CUDA and OpenCL workers do. Without it Mac node 1 and the Mac app cannot mine v3 (gates G1 and G4).
PC 1: the AMD-proving small fixture is running; Ember Tune's 8-minute CPU-only build takes the next CPU gap, its 30-minute both-cards run is job 5.
20:43. Consequences round 2 (C18 to C20). C18: "near parity per pound" was wrong and is struck everywhere; read-width.md section 4.1 gives the 5090 at 2.2x the 9070 XT per pound at list (0.072 against 0.032 MH/s per pound, approximate), 4.9x per watt, 7.5x in rate; the level 3 page carries those. C19 (to ca2-mixer, already sent by the reviewer): the x4 verifier cost in IBD minutes over the 108,000-header pruning window per tier (8.6 min against 1.1 on one M5 Max core at the top of the range), pool shares per core per second, a scaled 2019-class figure (approximate), and the 10 ms gate margin left for 3.0 go into mixer-x4.md; the seeds' header-verify load goes into the testnet go checklist. C20 (to ca2-epoch, already sent): both rig miners run --exit-on-seed-change and re-export on exit 42, so a 10-minute epoch restarts every card's miner six times an hour and the Mac fleet's prepare pause goes from 35 s to 3.5 min an hour; the epoch-length document gets a per-tier restart-cost row and the compile-ahead margin against the VDF at 600 DAA s, and the rig installer drops the exit-42 path for prepare-ahead before any short epoch can be drawn (next-cut list).
## 20:43 the Metal worker's v3 path; the app flag; the seam
Correction to the 20:42 entry: igneum-bench --serve DOES regenerate every program in Swift (serveProgram calls generateProgramV2), so a v3 line could not be trusted blind. ca2-node added servePackProgram in main.swift: a `prepare <e> <d> <dir> class=v3 era=<hex>` line compiles program_bound.metal from the pack the miner wrote (--prepare-packs) after the packfile.h checks in Swift (generator 2 or 3, class against generator, seed bytes against the line, IGNEUM_SEEDW_INIT against attempt_words, class and era against the line); the program store keys on (seed, class, era); a v3 job with no resident pack answers `need` + `error ... program class mismatch`; v2 lines unchanged. A pack program never races variants (the Mac loses the variant race on v3 epochs; its cost is the race's gain, from the miner-perf entry, to be quoted). Integration item: the Mac app and Mac node 1's miner command must pass --prepare-packs, else the first v3 epoch on the Mac worker ends in `need` lines; check app/igneum-app/src/engine.rs. The gate network's Mac miner is the CPU miner (igneum-miner --engine igneum-pow), unaffected.
The seam the node relies on (kept by every branch): Epoch::from_chain_seeds(epoch, day, era, class, label), Epoch::chain_program(epoch, era, class, label), Epoch::chain_dataset(day, class), generate_from_seed_bytes_program_class(label, seed, class, era), ProgramClass::{load_class, generator_version, from_generator, name, parse}, V3_CLASS, Program::era_bytes, packcheck::verify_pack_dir_chain.
Mac build queue: the measure lock has been held by a packbench run since 20:31Z with three build slots held and five builds waiting; the node's cargo check and the swiftc recompile wait behind it. This is the lock working as designed; it sets the pace of the gates tonight.
## 20:45 the 9070 XT has dropped off PC 1's bus; app restart facts; the era package ETA
The AMD sweep agent (a01dcb34ae16d867c) reports from a read-only probe at about 20:40 UTC: Get-PnpDevice lists only the integrated "AMD Radeon(TM) Graphics" (gfx1036) and the RTX 5090; the app's AMD worker now mines the gfx1036 at 3.12 MH/s; the 9070 XT is absent from PnP (the eGPU link: the Sonnet box or the USB4 router; earlier today it went Code 43 and came back after a driver reinstall and reboot). A 10-second rescan probe is granted (pnputil /scan-devices, the USB4 router status). the project lead is asleep and is not woken. Consequence if the card stays absent: gate G1 (bit-exact v3 on all three cards) and the 9070 XT rows of the era and hot-table tables cannot be taken tonight; the AMD-vendor stand-in available is the gfx1036 (RDNA 2, AMD OpenCL 3683.0, 3 MH/s), which ran the version 1 and version 2 conformance; whether it satisfies G1 for the devnet publish is asked of the coordinator. Every 9070 XT row taken before 20:40 (readwidth, dot4) stands.
App restart facts from the log intake (`node tools/logs.mjs`, 20:45): PC 1's app run is win-ae432dc7-20261005-190232 (started 19:02:32, no restart since), so no PC 1 measurement tonight straddled an app restart; PC 2's run is win-1ccfe586-20261005-200114 (started 20:01:14, before the readwidth 5090 round at 20:09). The 0.3.10 manifest is still unpublished (the shipper's CI is queued); the "jobs folder cleared by the 0.3.10 update" reading was wrong: the folder is cleared by fetch jobs.
The --prepare-packs item of 20:43 is resolved: app/igneum-app/src/engine.rs line 1225 passes it on master and on the 0.3.10 tree.
Era package: 10 to 15 minutes away (the pack-loop packfile merged over readwidth's; the OpenCL verification and the mingw rebuild queued behind three held build slots); zip ~/Desktop/igneum-ca2-era-pc1.zip, fetch id fetch-ca2-era-20261005, one playbook relay/playbooks/ca2-era-pc1.ps1 doing both cards (about 6 to 10 min). Widths pinned at 4 B in every era pack (512 B per hash), windows identical across the six packs, so the six-era spread isolates stride plus interleave. The CPU-only proving fixture holds PC 1 until about 21:15 to 21:30; the hot-table job is not ready either, so the order stays era, then hot table.
## 20:46 ruling on G1; hardware event recorded
Ruling (coordinator): gfx1036 satisfies the AMD vendor for gate G1 tonight (a compiler-and-ISA property; it carried the v1 and v2 conformance); the 9070 XT's hash-rate and power rows are owed and taken when the link is back; every 9070 XT row before 20:40 UTC stands. Nobody is woken, PC 1's app is not restarted. The event is in the rollout plan section 7b (hardware events) for the morning summary: the second eGPU link fault today (Code 43 at install, a bus drop at about 20:40); the project lead reseats the USB4 cable and the eGPU power; the 0.3.10 hot-plug code shows "removed" and picks the card up without a restart. The publish proceeds when every other gate is green.
## 20:47 the Sonnet box is off the link; PC 1 facts corrected; the queue after the era job
Rescan at 20:45:34Z (relay probe #203, 10 s): the 9070 XT stays absent after pnputil /scan-devices; the USB4 list shows only the host and root routers, the Sonnet Breakaway Box 850T5 router present at 17:18Z is gone: the box is off the link, not just the card. Job 4 (the 9070 XT sweep) is dropped, its rows owed with this reason and time. The era and hot-table PC jobs run their AMD half on the gfx1036 for bit-exactness only (the G1 ruling); their 9070 XT hash-rate and probe rows are owed.
Correction to the 20:45 entry: PC 1's app is 0.3.9 (file 15:47:20Z) and its process started at 20:01:14Z (pid 12340), a restart, not a 0.3.10 install; the log intake's run id dates the log file, not the process. Both PCs restarted at about 20:01Z, before every readwidth PC job (from 20:02:51Z) and the dot4 probe (20:27Z), so no measurement tonight straddled a restart. 0.3.10 is still unpublished.
PC 1 queue now: (1) the AMD-proving small fixture (running, release expected 21:15 to 21:30), (2) the era job (both halves, about 6 to 10 min), (3) the 5090 power-limit sweep (575 / 460 / 400 / 400 W, 90 s each, cap restored to 431 W, about 8 min; SM and memory clocks in the RESULT lines), (4) the hot-table job, (5) the reproducible benchmark (5 min), (6) Ember Tune's 8-minute build in a CPU gap then its 30-minute both-cards run, (7) the AMD-proving S_p shard (up to 90 min, CPU only).
## 20:49 mixer construction written; the measure lock cleared; PC 2 and the merged-tree suites
ca2-mixer (af345b1e2c541ffbb), no commit yet (lands when the crate tests and the v2 pack diff are green): LoadClass gains mixer_mult (1 or 4) and growth; LoadClass::MX4 = v2 loads, mixer x4, growth on, no width roll, so its program stream is version 2's draw for draw; memhard::Shape { mixer_mult, cache_log2_words } in MixParams; derive_items applies the mixer with keys round_key(r x m + j), j in 0..m, before each of the 8 reads and round_key(8m + j) after; Cache::fill_log2; growth_doublings(d) = ilog2(1 + d / 1460) (doublings at years 4, 12, 28, 60), cache_log2_words(d) = 26 + doublings, dataset_log2_words capped at 32; d = 1 on the devnet pack keeps 2^26 and 2^28. Emitters emit the m-loop only when m > 1 (v2 text byte for byte otherwise); program.h carries IGNEUM_MIXER_MULT and IGNEUM_CACHE_GROWTH. Seam addition (additive): Epoch::chain_dataset_day(day_bytes, class, days_since_genesis, genesis_dataset_log2); the node agent was told to wire the genesis day index from Params.genesis.timestamp and a pow_genesis_dataset_log2 field (28 on the devnet, in the digest). V3_CLASS becomes LoadClass::MX4 composed with the era and hot fields at integration. Numbers follow the lock.
The Mac measure lock: the packbench loop was the readwidth agent's (Metal currentAllocatedSize at 32 and 128 KiB scratch for the consequences reviewer's C12, not a decided row); it stopped the loop at about 20:52, so the queued builds (the node's cargo check, the mixer's tests, the epoch and mixer measurements) proceed.
PC 2: spcurve-stopped-pc2-pv1b closed 20:47:46Z (the card alone: an empty shard 13,875 MiB in 2.1 s; a v1 shard 20,435 MiB, 4.2 s; 2.25 M pgas 28,371 MiB, 6.6 s; 4.5 M 28,307 MiB, 8.5 s; the prototype 6.75 M 28,275 MiB, 11.2 s: the peak plateaus at 28.3 GB from 20 M cycles up); spcurve-miner-pc2-pv1 (the same with the miner on, about 5 min) runs now; PC 2 is released after it. The proving fork tip is 3203c8d0 (eb32c645 plus the pool test's field and N = 8), app 440fd59; its unit tests ran on the Mac (consensus-core 13, exec 8, flows, 0 failed); its harness runs on the Mac. Decision: ONE PC 2 suite job on the merged tree (ca2-v3-node plus 3203c8d0) once the ca2 node branch is committed, covering both halves of 0.3.11.
## 20:50 the measure lock holder is a prover measurement; a lock-status defect
Correction to the 20:50 entry above: the measure lock has been held since 20:31Z by pid 45000, a `with-lock.sh measure` of igneum-wt-agg-cost's igneum-prove-host (the aggregation-cost agent's prover measurement, 18 minutes so far), not by the readwidth packbench; `with-lock.sh status` prints the LAST WRITER's command text, not the holder's, which is why it named the w4 run. Defect for the next cut (tools/lock/with-lock.sh: the status line must read the holder's pid and command, not the last writer's; the class of CLAUDE.md's watcher rule). Builds proceed in the three build slots (build-0 taken at 20:49:51 after an 885-s wait); GPU measurements (the mixer's verifier and build timings, the epoch compile-ahead, the Mac bit-exactness runs under `run` are not blocked) queue behind the prover measurement. Readwidth head is 30ff674 (per-watt rows for the consequences reviewer; e752fc7 stays the table commit).
## 20:51 the number-free public copy is applied on ca2-coord (0ad70ba and the next commit)
site/litepaper.html: the Mining section's "bound by memory bandwidth" is corrected to "waits on memory latency, not on maths or bandwidth"; the "Everything above is automatic" paragraph is replaced by the level 2 three ideas and the fourth paragraph with a link to the numbers page; the vs RandomX rows "Changes over time" and "Dataset" and the Hardware paragraph carry the step schedule (years 4, 12, 28; 4 GB about four years, 8 GB about twelve). site/index.html: the Memory row and the Mine card carry the step schedule ("2 GB at genesis, doubling at years 4, 12 and 28"; "any 4 GB card at launch, 8 GB from year 4"). NOT yet applied, because they carry the chip number: the level 1 sentence in the hero and the abstract ("a custom chip gains under 2x") and the limits bullet "A chip is impossible"; they wait for the combined chip row from docs/analysis/chip-model-v3.md (if 1.8x: "under 2x" with the margin stated as thin; else qualified). The bench page's Counter ASIC section (level 3) waits for the final table. docs/evidence.md's card-lifetime row (designed) is still to add.
## 20:51 the node builds every day cache through chain_dataset_day
ca2-v3 6c75dad (node agent): Epoch::chain_dataset_day(day_bytes, class, days_since_genesis, genesis_dataset_log2) with a placeholder body (the mixer branch fills growth_doublings under that signature), verify::days_since_genesis; the Metal worker takes v3 from a pack (compiled). Fork (uncommitted, in the cargo check holding build-1 since 20:50Z): Params::pow_genesis_dataset_log2 (28 on every network, in the digest in its own statement), Params::genesis_day_index(), install_pow_genesis in the daemon after the class switch, the engine's build_day through chain_dataset_day, the pack export through the same build_day, PowEpochInfo / RPC / proto fields 17 and 18 (genesis_day_index, genesis_dataset_log2; an old node's 0 reads as 28), override-60x.json and the redteam override carry pow_genesis_dataset_log2 28. Next: the check result, then "ready for PC 2" with the fork commit. evidence.md row 22 (card lifetime, designed) added on ca2-coord (3dc29fe).
## 20:53 spec text applied for the decided layers (ca2-coord 9b1f849)
docs/spec/01-lottery-hash.md: 1.12 carries epoch_len (the ladder 600 to 7,200, 90% signal at a day boundary, T_epoch and the lead fixed) with 3,600 unchanged on every network; 1.13.1 gains the mixer_mult row (4 under class v3) and the epoch_len row with the signal rule, the FPGA threat and the 600-s floor, and a pointer to era-layout.md for the layer 4 and 8 rows; 1.13.2 carries reserve family R1 = mm8 (uint8, W_new 4, unlock era 4 or 90% signal, the edge vectors, the native paths) and the emulation rule (8x per op, 5% hash-rate cap); 1.13.3 carries the cache growth rule (option C, growth_doublings(d) = floor(log2(1 + d / 1,460)), 256 / 512 / 1,024 MiB at genesis / year 4 / year 12, fill 0.2 / 0.4 / 0.8 s), the step mapping (b) as recommended, the shipped-density reason, and the cache freed after the daily build. docs/spec/04-seeds-and-vdf.md 4.3: the lead and T_epoch fixed at every epoch length. Still to land in the spec from the branches: 1.8.5 (the mixer x4 form, from mixer-x4.md), the 1.13.1 rows for stride, interleave and the window (era-layout.md), 1.5 and 1.8 for the hot table in the added form (hot-table.md), 1.17 and 1.15 for the v3 vectors and the conformance runs, 1.4.5 and 1.4.6 for generator 3 and the class in the pack.
## 20:56 PC 2 released; the proving fork tip for the merged tree
PC 2 is free (the proving agent's last job closed; the live prover is back on). Proving fork tip ece42979 on 21d4c73c (N = 8, the digest test edit), app 440fd59 or later; its suites ride with the ca2 node suites on the merged tree; its harness on the final tree is running on the Mac. The S_p curve with the miner on the card (PC 2's 5090): empty shard 15,585 MiB 7.5 s; the adopted v1 shard (30,000 pgas, 4.7 M cycles) 22,210 MiB 13.2 s (20,435 MiB, 4.2 s alone); 2.25 M pgas 30,049 MiB 17.9 s; 4.5 M 29,954 MiB 26.3 s; the prototype shard 30,083 MiB 33.3 s. Tiers as the proving agent published them: 32 GB mines and proves today, 24 GB from the fee switch (2.3 GB spare on the adopted shard), 16 GB empty shards only, 12 GB nothing on this build (D2 to the project lead).
PC 2 queue: the ca2 node suites on the merged tree (ca2-v3-node + ece42979) as soon as the node agent sends "ready for PC 2"; nothing else is queued on PC 2.
## 20:57 the mixer construction is committed on ca2-v3's base
ca2-mixer 0fc0ad1 (rebased onto ca2-v3 6c75dad): LoadClass::MX4 = V3_CLASS (v2 loads, mixer x4, the growth rule; a class v3 program is the v2 program of its seed instruction for instruction, generator 3 in its id); chain_dataset_day has its real body (Shape::for_class_day: cache 2^cache_log2_words(d), dataset 2^dataset_log2_words(D_0, d)); 44 lib + 11 pack tests green, the two pinned v2 packs byte for byte. Next from it: --program-class v3 / --era-hex on the CLI, the pinned v3 packs mx4-genesis and mx4-devnet-epoch0 (era = the genesis-hash stand-in), the design and 1.8.5 spec text, the chip-model row; bit-exactness runs under the run lock now; the verifier and build timings wait for the measure lock (held by a live prover measurement from another worktree, pid 45000, 25 min in at 20:56). The node agent merges ca2-mixer before its fast-time gate, so the gate runs the real construction minus the era and hot fields.
20:58. PC 1: cpu-prove-pc1-small2 running since 20:54:02 (CPU only). PC 2: a fetch job from the aggregation-cost agent (job-fetch-prove-aggcost, 20:55:39) landed after the proving agent's release, so PC 2 is NOT idle for the ca2 suites until that agent's run closes; the suite publish checks `node tools/jobs.mjs status` for an idle PC 2 first. Readiness in hand: the repro benchmark (8 min, 5090 then gfx1036) waits for a gap; the 5090 power sweep (8 min) follows the era job; Ember Tune's build (8 min, CPU) and run (30 min, both cards) follow; the S_p CPU shard (up to 90 min) is last.
## 21:00 PC 1 released by the CPU fixture; the repro run has it; PC 2 to the aggregation-cost agent
cpu-prove-pc1-small2 finished 20:59:49Z: the SP1 CPU prover on PC 1 with the miners running: block-56-transfers-3shards shard 0 (200 pgas) 312 s wall, peak RSS 29.5 GB, 978% CPU; block-78-increment 322 s, 30.5 GB; the 5090 untouched (89% mean). The S_p shard job is DROPPED tonight: 312 s for 315 k cycles extrapolates the 60.8 M-cycle shard to many hours of every core and over 30 GB (approximate), which answers the CPU-fallback question (not viable for S_p shards; viable for empty or tiny shards only). "go PC 1" given to the reproducible benchmark (the 5090 then the gfx1036, about 8 min); the era job follows it, then the 5090 power sweep, then the hot table, then Ember Tune's build and run. "go PC 2" given to the aggregation-cost agent (20 min, GPU proving with the miner on then paused, prover restored); then the ca2 node suites, then the repro run's PC 2 slot (10 min).
21:01. GitHub Actions is in a major outage (six queued runs since 19:26Z, none acquired); the coordinator gave the 0.3.10 shipper the fallback at 21:00Z: build the Windows installer on PC 1 (MSVC window host, the payload under Git Bash, Inno Setup; CPU only, about 15 min). PC 1 order now: the repro run (until about 21:09), then the 0.3.10 installer build (the fleet's release, ahead of every measurement), then the era job, the 5090 power sweep, the hot table, Ember Tune. Any measurement that straddles the build window is re-run.
21:02. The Mac measure lock, found by `lsof`: the recorded holder pid 43916 is dead; the files are held open by two WAITERS, the readwidth agent's re-queued footprint loop (pid 78893, holding the measure and build files 11 min, waiting for the three build slots, which cargo tests keep re-acquiring: a convoy) and the epoch agent's compile-ahead measurement (pid 78476, waiting behind it). The readwidth agent is asked to kill 78893; the epoch measurement then runs when the build slots drain. Two defects for the next cut (one task filed): the status line shows the last writer, not the holder; a measure waiter can hold the master lock while build slots keep being granted to new builds, so a measurement can wait indefinitely under a steady stream of cargo tests.
## 21:04 proving v1 handoff received; the AMD-proving line on the site; amd-prove merged
Proving v1 for 0.3.11 (rollout plan 8a): fork ece42979 on 21d4c73c, harness PASSED (21 checks) in 244.4 s at 20:56:45Z on the final fork tree, override fields and the mixed-fleet rule recorded; the app's final hash follows its gate tests (90d3299 before it). Branch amd-prove (f1d7a7d) merged into ca2-coord (the append-only bench-log conflict kept both entries); its finding: no zkVM proves on an AMD GPU as of 5 October 2026 (SP1 CPU and CUDA; RISC Zero and ICICLE add Metal; nothing for AMD), the CPU fallback is about 5 minutes per small shard at 30 GB RSS, not a tier. The public line is applied on ca2-coord (1c8439f) with the proving agent's measured tiers in place of the doc's 16 and 20 GB: "Proving needs an NVIDIA card with 24 GB or more (32 GB until the fee switch of 6 October 2026; from it a 24 GB card mines and proves on the same card: 22.2 GB peak with the miner on). AMD and Apple cards mine. A prover for them lands when a zkVM ships one." It replaces "the card mines and proves" on the index, the litepaper's vs RandomX row and proving section, and the miner page (title, meta, hero, feature). Left as it was: the app's Proving tile text (the proving agent's).
## 21:05 the node merge for 0.3.11 is done; the hot-table package is ready; unproven_daa 10 at fast time
ca2-v3-node: 2e464e81 (the class switch, the era stand-in, the template, the miner) and the merge of proving-v1 ece42979 = ba43cf0f. One conflict, params.rs's digest-test edits array, resolved by keeping both sides (14 entries); every other shared hunk auto-merged as separate blocks. The merged tree is in cargo check (with igneum-exec and kaspa-p2p-flows); "ready for PC 2" follows with ba43cf0f once it and the Mac pow and consensus-core tests are green. Main repo ca2-v3: ca2-mixer fast-forwarded (66eeba3), then 43ca289 (the fast-time and redteam overrides carry the proving v1 fields and pow_genesis_dataset_log2 28; class-v3.mjs prints the dataset build ms per epoch). proving_v1_unproven_daa is a DAA clock (exec/src/proving.rs segment_status), so 10 at fast time is right (the proving agent confirms; its harness passes --unproven itself). The PC 2 suite command is the node agent's final shape (75-minute budget, six node crates plus the app tests), published from the ca2-v3 worktree when PC 2 is idle (the aggregation-cost job closes at about 21:21).
ca2-cache 196db96 (rebased on ca2-v3 464d6e1, the added form): 47 + 13 tests; the three added packs bit-exact on Metal and Apple OpenCL (fingerprints at 2^20, base 0: hot32k4a afb700b2d997c847, hot64k4a ba214baa9c1a9e85, hot96k4a 29e1916aed6deff5; 96/96; hot table PASS); first-pass rates under load (not numbers): 25.7 / 23.9 / 22.9 MH/s against v2 27.7 (the probe predicts 0.96 / 0.94 / 0.94); the measured rows wait for the measure lock behind the epoch measurement. PC package: ~/Desktop/igneum-ca2-hot.zip sha256 bd49faa1c9d48024f49c615481faff5c68a4c09f0889dbaf009c208674d67b3f (workers 956c4ab3... and 32d3d343... from 196db96, eight packs), fetch-ca2-hot-20261005, playbooks ca2-hot-5090-bench.ps1 and ca2-hot-9070-bench.ps1 (gfx1036 fallback). Its go follows the 0.3.10 build on PC 1 unless the era package is there first.
## 21:11 the 9070 XT is back; PC 1 to the 0.3.10 build; the repro run failed at parse time
The repro run (run-repro-pc1-20261005, 21:04 to 21:09:24Z) switched every card off and on and restored them, and found the 9070 XT (gfx1201) ON the bus again (opencl:1; the app switched it off and on), so the era and hot-table jobs run their gfx1201 halves as planned and the G1 ruling's fallback is not needed unless the link drops again; recorded in the rollout plan's hardware events. The run produced no numbers: repro.ps1 failed with a PowerShell parse error on each card (MissingEndParenthesisInExpression), the second playbook tonight that passed no local parse (the Mac has no pwsh); the class fix ordered: every PowerShell playbook parses itself on the PC as its first step (System.Management.Automation.Language.Parser::ParseFile, errors printed, non-zero exit), the way the dot4 playbook gates its bash body with bash -n. "PC 1 is yours" given to the 0.3.10 shipper at 21:10 for the installer build (CPU only, about 15 min); then the hot-table and era jobs, the 5090 power sweep, the repro re-run (about 22:00), Ember Tune.
21:20. Proving v1 app branch final: 6dc686a on 5b0d54f (provedefault 6 of 6; the app's 0.3.11 inputs are complete on that side); fork stays ece42979 (merged into ca2-v3-node ba43cf0f). The 0.3.11 app tree = the 0.3.10 release tree 5b0d54f + proving-v1 6dc686a + whatever the app needs for v3 (the --prepare-packs flag is already there; the Metal worker change is in the worker, not the app); the 0.3.11 main tree = master + ca2-v3 (igneum-pow, workers, fast-time, docs) + ca2-coord (the plans, the spec, the site copy).
## 21:20 layer 9 complete (ca2-epoch 4300608, e95e8b5)
Design (reserve-only): epoch_len base 3,600, ladder 600 to 7,200, SIGNAL ONLY (the era stream consumes draw 8 and ignores it); day-anchored epochs so a change lands in days; VDF option A (T_epoch 600 s and the 1,200-s lead stay genesis constants: the program is known 600 s ahead at every length, grinding margin 300x; option B rejected at 50x and a 200-s-deep checkpoint); REF_WINDOW_V2 = min(600, L); the 144-s settle per step is 24% of a 600-s epoch. Floor 600 DAA s: the slowest compile-ahead is the variant race at 38 s on the Mac and the 5090, 6.3% of the epoch and inside the window.
| Card | Compile-ahead | Share of a 600-s epoch |
|---|---|---|
| M5 Max Metal, race off (measured 21:18 UTC: 15.9 / 17.7 / 20.4 ms over 10 fresh programs, pack 79 ms then 1 ms) | 0.5 s (1.8 s cold) | 0.1% |
| M5 Max, race on (M11) | 38 s | 6.3% |
| RTX 5090 NVRTC, race off (M11, hot-swap) | 1.0 s | 0.2% |
| RTX 5090, race on | 38 s | 6.3% |
| RX 9070 XT OpenCL | 0.31 s + compile owed | |
| Intel UHD OpenCL (M11) | 6.4 s | 1.1% |
| Radeon iGPU under load (M11) | 124 s with the dataset | 20.7% (needs per-day dataset reuse before any signal below the base) |
FPGA citations: PRflow (FPT 2019) 42 min typical, 160 min worst for a monolithic Vivado compile; Aldec hours on Virtex UltraScale; partial reconfiguration milliseconds per region (ICAP 400 MB/s) shortens the load, not the compile; at 600 s a per-program bitstream mines 0% of each epoch, 47% at 3,600. Riders: the race defaults off (M11: base wins on both cards; 6.3% at the floor); per-day dataset reuse in the workers for the iGPU tier. Owed: the 9070 XT clBuildProgram time, the difficulty settle at a 15% step and 600-s epochs in sim.py, the 3-bit signal encoding against Kaspa's version bits (spec 5.8).
## 21:20 the hot table in the added form, measured on the Mac: the big-die chip comes out ahead
ca2-cache (hot-table.md 6.2 to 6.4): M5 Max, 21:03 to 21:19 UTC, load average 7 to 14, Metal packbench 2^24 x 5 (GPU time) and Apple OpenCL --bench-pack; v2 in the same session 27.63 / 27.59 MH/s.
| Pack | Metal / OpenCL MH/s | g against v2 (probe predicted) | Fingerprint (2^20, base 0) | Verifier ms per warp (v2 0.602) | Fill, one core |
|---|---|---|---|---|---|
| hot32k4a | 25.76 / 25.72 | 0.93 (0.96) | 8a3414735db4523c | 0.631 | 21.7 ms |
| hot64k4a | 23.92 / 23.87 | 0.87 (0.94) | 45668f34105f6307 | 0.609 | 43.3 ms |
| hot96k4a | 22.92 / 22.88 | 0.83 (0.93) | af763997dfee4c82 | 0.614 | 64.9 ms |
Reading: on Apple the added form costs 7 to 17% of the rate for four extra loads per iteration, more than the probe predicts as the table grows; only the 32 MiB table is near free. Chip arithmetic with these g: a chip serving H from DRAM keeps 0.86 / 0.92 / 0.96 of its gain; a 100 mm^2 die with the SRAM keeps 0.96 / 0.92 / 0.88; a 750 mm^2 die with the SRAM comes out 6 to 15% AHEAD, because the GPU pays the hits in rate and a big die pays them in 1.6 to 4.8% of area. So on the Mac's numbers layer 5 does not pass its own test; the decision waits for the 5090 (96 MiB L2) and 9070 XT (64 MB Infinity Cache) rows, where the hits may be near free (g close to 1). Rule for the decision: layer 5 goes into v3 only if, on every card we own, g is at or above 0.97 at the chosen size AND the on-die-cache chip row (chip-model-v3.md) moves down with it; otherwise layer 5 is out of v3 and stays a measured option for 3.0.
21:21. PC 2: agg-cost-pc2-1 closes at about 21:25Z (its own-miner phases ran the iGPU miner by a script fault; the curve is unmeasured); "go PC 2" given for agg-cost-pc2-2 (about 17 min, to 21:44Z), release due by 21:45Z; the 0.3.11 suites on ba43cf0f take PC 2 next (the node agent's Mac suites, release build and fast-time gate are in flight). PC 1: the 0.3.10 installer build (from 21:10); the hot-table job (package in hand) goes the moment the shipper reports the build closed; the era package is still being built.
21:22. Ember Tune (f9bf552 on ember-tune) is queued: its Windows app build (8 min, CPU) in the first gap after the installer build, its baseline run on the 5090 (8 min, no prompt; the 9070 XT if it can be taken without one) after the repro re-run. PC 1 order: the 0.3.10 installer build (running), the hot-table job, the era job, the 5090 power sweep, Ember's build in the first CPU gap, the repro re-run (about 22:00), Ember's run.
21:22. PC 1: the 0.3.10 installer build job (job-fb-installer-pc1) FAILED at 21:20:05Z, exit 2 after 2 s (the script, not the build); the shipper is asked to republish within 5 minutes or yield PC 1 to the hot-table measurement (10 min) and follow it. PC 2: agg-cost-pc2-1 still running at 21:21.
21:23. The shipper republished the installer build as fb-installer-pc1-2 (the first exit 2 was Test-Path on a \\wsl$ root path refused to the non-elevated session; the engine is now copied out with wsl -u root), cap 25 min, so PC 1 is the build's until about 21:50; the AMD presence probe (10 s, nothing held) runs beside it. Then: the hot-table job, the era job, the 5090 power sweep, Ember's build and fetches, the repro re-run, the AMD sweep (26 min), Ember's run (25 min).
## 21:23 the era package is ready; the V3_CLASS composition rule
ca2-era PC 1 package: ~/Desktop/igneum-ca2-era-pc1.zip sha256 f26f997602d94b0a408a6974ee484a7b3249ce18dc91dc00783aa9754e5ff040 (workers 5dd3bc16... and 9615efb3... from ca2-era on ca2-v3 464d6e1 with the pack-loop packfile.h merge and the host.c mh_word fix; packs v2 and era-0 to era-5, the six as class v3 chain packs, generator 3, program id 6b02c7c49eb126bd shared, width 4 B, the same devnet epoch seed and day); fetch-ca2-era-20261005; relay/playbooks/ca2-era-pc1.ps1 (the 5090 then gfx1201, gfx1036 fallback; 6 to 10 min). Mac bit-exactness on these packs: Metal 6/6, Apple OpenCL 6/6 (same fingerprints), CUDA CPU emulation 6/6. Its go follows the hot-table job, about 22:00.
Composition rule for the integration (three branches define V3_CLASS): V3_CLASS = LoadClass::MX4's fields (v2 loads, mixer_mult 4, growth on) + era: None (drawn per program inside generate_from_seed_bytes_program_class from the era bytes, LoadClass::era(V3_CLASS, era, &V3_ALLOWED)) + hot: None until the PC rows decide layer 5; the layout rides with the program (program.class.layout() in the interpreter and Epoch::dataset_word), so one day cache (chain_dataset_day) serves every era. The era agent rebases onto ca2-v3 HEAD with that literal; the node agent takes it into the integration tree. The era stream is seeded from "igneum-era/" || E_n (the index dropped: E_n commits to n through the VDF input); the spec text in era-layout.md says so.
21:24. The 9070 XT dropped off PC 1's bus again (probe #205 at 21:22:59Z: no Sonnet or USB4 router device; the third drop today; rollout plan 7b updated: the link is flapping). The era and hot-table jobs run their gfx1036 fallback unless the card is present at run time; the AMD sweep slot is conditional on a presence probe at 22:00. A probe reading of "miners off at 0.0 MH/s" was wrong: the app log shows the 5090 at 124.5 MH/s through 21:23:26Z; the AMD agent fixes its probe's precondition. The 0.3.10 installer build fb-installer-pc1-2 runs on PC 1 (cap 25 min from 21:23).
## 21:25 the mixer x4 construction is bit-exact on the Mac; the chip row reads 1.84x (thin)
ca2-mixer commits: 0fc0ad1 (construction, V3_CLASS = mx4, chain_dataset_day body), 66eeba3 (the pinned v3 packs mx4-genesis and mx4-devnet-epoch0, --program-class v3 / --era-hex), e4c04a7 (tests/mixer.rs: 200-program v3 fuzz, stats, edges, determinism; scratch.rs cherry-picked, 7 of 7), 7ce8d1e (docs/analysis/chip-model-v3.md). Spec text for 1.8.5 and 1.13.3 in docs/plans/mixer-x4.md section 2: the multiplied mixer with keys (r m + j + 1) x 0x9E3779B9; option C as doublings(d) = floor(log2(1 + d / 1460)); the day table with the verifier fill per step.
Bit-exactness (run lock): both v3 packs on Metal and Apple OpenCL, 3/3 standalone and 3/3 in batch, 96 of 96 lanes, dataset head / MASK / 64 samples PASS, one fingerprint per pack across both harnesses (6f48d5a2aa0dbe5f, 73caaebb28e808fe); the 200-pack v3 fuzz on Metal 200 of 200, every tenth on Apple OpenCL 20 of 20; CPU 44 lib, 12 packs, 7 scratch, 4 mixer tests; v2 exports IDENTICAL. Indicative (run lock, not a number): the Metal 1 GiB build at x4 30.2 ms (mx4-genesis) and 21.7 ms (mx4-devnet); the measured verifier and build timings wait on the measure lock.
The chip row (chip-model-v3.md), v3 at x4: 599,040 ops per hash; the on-die-cache recompute chip at 50 T op/s does 83.5 MH/s, 0.61x bare against the 5090's 136.1, 1.84x with the 3x fixed-function factor, 1.53x with the 128 mm^2 mirror deducted at equal silicon. With the hot table in the ADDED form at the Mac's g: 1.98x (32 MiB) and 2.12x (64 MiB) at equal budget, 1.60x / 1.67x with the SRAM deducted: the added hot table costs the card and not this chip, so it moves the row the WRONG way on the Mac's numbers. The claim holds "under 2x" on the equal-silicon convention, and on the equal-budget one only without the hot table; thin everywhere (a 3.3x factor or 10% on the budget reads 2.0x). Next lever: x8 (0.31x bare, 0.92x with the factor).
Consequence for layer 5: unless the 5090 and 9070 XT rows show g at or above 0.97 (the hits near free), the hot table stays OUT of v3 tonight (the rule in the 21:20 entry) and the public level 3 names it as a measured option, not a lever.
21:25. "ready for PC 2" from the node agent: fork ca2-v3-node 79bd8e10, Mac checks and tests green (rollout plan G6 row); the expected 0.3.11 digest with the two new fields at never is c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c. The coordinator publishes the suite job from the ca2-v3 worktree at 21:45 when the aggregation-cost agent releases PC 2, on the worktree's tip at that moment (the era and cache merges go in first if the era commit arrives in time).
21:27. The installer build's second attempt failed in 4 s on a path (the engine is not under /root/igneum-build/app/...); attempt 3 (fb-installer-pc1-3) is publishing with a find-based path; rule: if it fails inside 5 minutes, the hot-table measurement takes PC 1 before attempt 4. The shipper's jobs started no second app instance (the failed attempts exited before any exe ran); its post-build listing of igneum-app.exe processes settles the "off" reading.
## 21:27 decision rule for the mixer: x8 beside x4
Delegated (coordinator, the project lead's "as strong as the measurements allow"): the mixer agent builds LoadClass::MX8 beside MX4, exports mx8-genesis and mx8-devnet-epoch0, runs the Mac bit-exactness, and measures v2, x4 and x8 in one measure-lock session (verifier ms per warp on one core, avg of 50 and worst cold; the 256 MiB fill; the Metal 1 GiB build); the 5090 and AMD daily-build times come from a 2-minute prepare job on PC 1 after the era job. x8 goes into v3 if the per-warp verify stays under 10 ms on one core AND the daily build stays under 1 s on every card we own; else x4 with the thin margin stated (1.84x with the factor) and x8 named as the next lever (0.92x with the factor, from the m16 table). The vectors are re-cut once after the choice. The hot-table rule stands (into v3 only at g >= 0.97 on both PC cards).
21:28. Installer attempt 3 failed at 21:27:20Z (exit 2 in 4 s: an inline bash -c string lost a quote through PowerShell; the fix is the 0.3.6 cut's: the WSL part as a file run with bash <file>, plus a read-only path probe before attempt 4). By the rule, "go PC 1" went to the hot-table job at 21:28 (10 min). PC 1 order from here: the hot-table job, the shipper's path probe and attempt 4 (about 16 min), the era job, the 5090 power sweep, the mixer daily-build job (2 min), Ember's build, the repro re-run, the AMD sweep (conditional), Ember's run.
21:31. Consequences round 3. C23: the x8 table gains a gfx1036 row (the integrated tier builds the dataset per PREPARE: 7 to 12 s at x1, 55 to 124 s under load, so 28 to 48 s an hour at x4 and 56 to 96 s at x8, every boundary missed) and a scaled 8 GB-class row; the "under 1 s" rule applies to the discrete cards' daily build; for the integrated tier either per-day dataset reuse in the three workers lands with v3 (the node agent is asked whether it is bounded tonight) or the level 3 page says the iGPU tier mines v3 with a restart per epoch; recorded with the x4/x8 choice. C24: two inline bash bodies lost a quote through PowerShell tonight (amd-prove's awk at 20:44, the installer's bash -c at 21:27); the reviewer's sub-agent adds a repo-wide CI check (every inline bash body through bash -n, an unextractable one fails CI) and the convention line in packaging/README-ship.md; no collision with the PC 1 queue.
## 21:31 GitHub's runner recovered: the 0.3.10 rollout starts; the restart rule for every PC job
The 0.3.10 Windows run 37374158235 went green at 21:30:29Z on the release tree; the three fb-installer jobs are done and expired; nothing more of the shipper's touches PC 1. The rollout runs now (manifest, update-now, hand nodes, seed, digest sweep): every app restarts once within the next quarter hour. Rules: the shipper is asked to hold PC 1's update-now until the hot-table job releases (about 21:40) and the era job starts only after PC 1's new STATUS line; the 0.3.11 suites publish on PC 2 only after PC 2's app shows 0.3.10 (a build job dies with the app); any measurement that straddles a restart is re-run; every PC job's RESULT lines carry the app version before and after.
## 21:32 per-day dataset reuse: Metal has it, CUDA and OpenCL do not (0.3.12); the gate re-runs after a script fix
Node agent: the Metal worker's ServeStore already keys datasets by day and programs by epoch (a prepare on a resident day builds the program only); the CUDA worker.cpp and OpenCL host.c bundle program, cache, dataset and the self-test in one Pair, and splitting a Day object out touches buffer ownership, releasePair, the prepare thread and the self-test in both: over an hour, not shipped untested tonight; first item after the publish (0.3.12; next-cut list). Decision recorded (C23): the level 3 page states that the integrated tier on the one-click workers mines v3 with a restart per epoch (public copy and rollout plan updated).
Gate G4: the first fast-time run failed at node start on the script, not the node: JSON.parse turned a never height (18446744073709551615) into 1.8446744073709552e+19 and the node refused the override; fixed as text merging in class-v3.mjs and simnet.mjs (d5ff532; no other script in tools/, infra/ or sim/ has the shape). The gate runs again on the mixer-x4 class (binaries from 79bd8e10 + ca2-v3 66eeba3). Main-repo tip for the suite job title: d5ff532 (era and cache not yet merged).
21:34. PC 2: agg-cost-pc2-1 closed 21:25:11Z (done, miners and prover back on); agg-cost-pc2-2 went out at 21:33Z (a missed close), self-limited to 16.5 min, closes about 21:51Z; the 0.3.11 suites publish after it AND after PC 2's app shows the 0.3.10 STATUS line (the update-now goes to PC 2 now). PC 1: the hot-table job runs (release about 21:40); PC 1's update-now follows the release; the era job starts after PC 1's 0.3.10 STATUS line.
## 21:36 layer 5 measured on both PC cards: OUT of v3
Hot-table jobs on PC 1 (fetch 21:29:01Z; run-ca2-hot-5090 126 s, closed 21:31:38Z; run-ca2-hot-9070 247 s, closed 21:35:39Z; both cards restored; the 9070 XT WAS on the bus, full path; app 0.3.9 throughout, no straddle). All eight packs bit-exact on the 5090 and the 9070 XT with the Mac's fingerprints.
| Pack | RTX 5090 MH/s (v2 136.1) | RX 9070 XT MH/s (v2 18.15) | M5 Max g |
|---|---|---|---|
| hot32k4 (replaced) | 146.6 (x1.08) | 19.79 (x1.09) | x1.22 |
| hot64k4 | 140.8 (x1.03) | 18.73 (x1.03) | x1.12 |
| hot96k4 | 138.5 (x1.02) | 18.33 (x1.01) | x1.05 |
| hot64k2 | 137.5 (x1.01) | 18.17 (x1.00) | x1.00 |
| hot64k8 | 163.6 (x1.20) | 22.32 (x1.23) | x1.71 |
| hot32k4a (added) | 118.7 (0.87) | 15.27 (0.84) | 0.93 |
| hot64k4a | 115.4 (0.85) | 14.62 (0.81) | 0.87 |
| hot96k4a | 114.4 (0.84) | 14.56 (0.80) | 0.83 |
Decision (the 0.97 rule): layer 5 is OUT of v3. Neither card keeps even the 32 MiB table resident while the 1 GiB dataset streams (the replaced form gains 1.02 to 1.08x at k = 4 against an ideal 1.33x), and the added form costs 13 to 20%; the chip row moves the wrong way with it. Layer 5 stays a measured option for 3.0 (a table small enough to stay resident, or a different access pattern). The probe rows and the writeup follow on ca2-cache. PC 1 is released to the shipper for PC 1's update-now; the era job starts after PC 1's 0.3.10 STATUS line.
21:38. ca2-cache final: 2de19e5 (nine commits from 55e285c, on ca2-v3 464d6e1); hot-table.md carries the probe rows for all three cards (5090 112.6 G loads/s at 32 / 64 / 96 MiB inside its L2 against 17.6 at 1 GiB; 9070 XT 9.88 / 9.47 / 8.18 / 2.43; M5 Max 21.7 / 12.8 / 12.3 / 3.50), the PC tables with g, the chip arithmetic at the measured g, the decision, the 3.0 note ("what would make it pay": a resident size found by a hash sweep below 32 MiB, k only with residency, a line-unit or streamed access shape) and the unverified list; the bench-log entry and two addenda carry the job ids and the worker sha256s. The probe promises full hits inside the 5090's L2 but the hash gets 2 to 8% at k = 4 because the streaming dataset evicts the table.
## 21:39 gate G4 run 1 PASS on the mixer-x4 class
Fast-time 3-node network, 21:33 to 21:38 UTC (fork 79bd8e10 + igneum-pow 66eeba3): the switch line on 3 of 3 nodes (active from epoch 3, DAA 150 rounded up to 180), templates class 2 then 3 from epoch 3, 181 blocks before and 124 after the boundary, program ids agree on all three miners (v2 e0 to e2, v3 e3 to e5), 0 rejected on miners and nodes, one sink on all three (082fd39ba65df2ff, 304/304/304), a new (day, class) cache 177 to 235 ms on one core. Main-repo tip bf04c56 (the doc, the summary JSON, the script's --connect fix). Run 2 on the composed class follows the era commit and the cache merge (rebuild about 10 min, gate 5 min). The x4 dataset build time comes from the mixer's PC prepare job (the CPU miner derives words from the cache).
## 21:40 x8 built and bit-exact beside x4; the mixer PC job retargeted to PC 1
ca2-mixer 504cae4 (LoadClass::MX8 "mx8", packs mx8-genesis and mx8-devnet-epoch0, the fuzz takes IGNEUM_MIXER_CLASS, playbooks with the x8 packs) and fe4e193 (mixer-x4.md per-tier build table and the x4/x8 rule; chip-model-v3.md with the mixer row as the headline, the layer 5 rows kept as measured not adopted with the PC g beside the Mac's; x8 rows 0.31x bare, 0.92x with the factor, 0.76x at equal silicon at year 0). x8 bit-exactness (run lock): both packs on Metal and Apple OpenCL 3/3 + 3/3, 96 of 96 lanes, one fingerprint per pack across both harnesses (7c28cfb06c5c65a9, bbb183f72692f840); 50-program x8 fuzz on Metal 50 of 50, every tenth on OpenCL 5 of 5. Indicative Mac builds (run lock): 30.0 ms at x8 against 30.2 at x4 (genesis pack), 22.0 against 21.7 (devnet pack): the Mac's build is latency-bound. The timing session (verifier v2 / x4 / x8, the fill, the build) is queued behind the measure lock. PC job: mixer-x4-pcjob.zip sha256 55a2913cb8790cd3b106dc3d0d29b6e2952b5a08378cd815935915ab82898a8e (v2 control plus the mx4 and mx8 genesis and devnet packs; a prepare per pack printing the worker's cache and dataset build ms); retargeted so both halves run on PC 1 (its 5090 and its AMD card), after the era job, about 22:05.
## 21:41 the mixer timing session: x8 passes the verifier half of the rule
M5 Max, one core, measure lock, 21:40:12 to 21:40:23 UTC, on a loaded box (load average 5.6 one-minute, 26 fifteen-minute: other agents' unlocked processes), so the absolute figures are about 2x the quiet 0.604 ms v2 baseline and the RATIOS are the measurement (two rounds, within 4%); a quiet-box re-run is owed for absolute numbers.
| Class | Verifier ms per warp, avg of 50 (round 1 / 2) | Worst cold unit | Ratio to v2 |
|---|---|---|---|
| v2 | 1.361 / 1.310 | 1.579 | 1 |
| x4 (genesis; devnet pack 1.923) | 1.956 / 1.923 | 2.043 | 1.45x |
| x8 (genesis; devnet pack 2.972) | 2.785 / 2.790 | 2.942 | 2.1x |
256 MiB cache fill on one core 172 to 175 ms. Metal 1 GiB build, GPU time: v2 21.0 ms (29.7 cold), x4 20.9 / 21.0, x8 21.9 / 21.9: the Mac's build is bound by the 8 dependent cache-line reads per item, not the arithmetic, so the "under 1 s on every discrete card" half of the rule is decided by the 5090 and 9070 XT rows of the mixer PC job (by the M16 arithmetic the 5090 is 54 ms at x4 and 107 ms at x8 if arithmetic-bound, 13.4 ms if latency-bound: far under 1 s either way). Verifier half: x8 passes with 7.1 ms of the 10 ms gate to spare on the loaded core (about 1.3 ms on a quiet core, approximate); x4 leaves 8.0 ms. Chip row at x8: 1,198,080 ops per hash, 41.7 MH/s, 0.31x bare, 0.92x with the 3x factor, 0.76x at equal silicon; x4 1.84x / 1.53x. C19 at these loaded figures: shares per core per second 735 / 511 / 358 (v2 / x4 / x8), a 22,000-member pool at one share per 10 s needs 3.0 / 4.3 / 6.1 cores; IBD over 108,000 headers on one core 2.4 / 3.5 / 5.0 min. Provisional choice under the rule: x8, confirmed when the PC build rows land (about 22:10).
## 21:41 the era draw passes the 5% rule on the Mac; the final era package; the merges for G4 run 2
ca2-era 9f98af2 (one commit on ca2-v3 HEAD; 48 lib + 17 integration tests; the pinned v2 packs byte-identical; mx4-genesis unchanged; mx4-devnet-epoch0 re-exported with the era inside the class). V3_CLASS = LoadClass { era: None, ..LoadClass::MX4 }; the era class is drawn inside generate_from_seed_bytes_program_class from the era bytes; chain_dataset_day and Epoch::dataset_word compose unchanged; an era program takes 11 draws per instruction. Mac (M5 Max, 21:38Z, Metal packbench 5 x 2^24, a loaded box): v2 27.68 MH/s; era-0 to era-5 28.58, 28.48, 28.35, 28.38, 28.49, 28.48: min 28.35, median 28.48, max 28.58, SPREAD 0.8% (under the 5% rule); 3/3 vectors and the in-batch vectors PASS on every pack; CPU verify 1.319 to 1.345 ms per warp against v2 1.334 in the same loaded run (quiet re-run owed). FINAL PC 1 package: ~/Desktop/igneum-ca2-era-pc1.zip sha256 f79c0607bb4187e3cf16fce3f533e7d525673d766d7edb799f27fd81af5dcee1 (workers 0fbfd50a... and 8c8caff7... from the merged tree; packs re-exported, attempt 0, program id 73bcbfe8ccf988f1 in all six with the era seed beside it); fetch-ca2-era-20261005; ca2-era-pc1.ps1. The 0.3.10 update-now reached every machine at 21:39:59Z (manifest live 21:33Z); the era job starts on PC 1's 0.3.10 STATUS line. The node agent merges 9f98af2 then ca2-cache 2de19e5 into ca2-v3 for G4 run 2 and the suites' tip.
21:42. ca2-v3 now carries the era draw: ca2-era's tip b105a55 (9f98af2 rebased onto the node agent's 88dafbc) fast-forwarded, no conflict; the suite job's main-repo tip is b105a55. ca2-cache 2de19e5 does NOT merge (it bases on 464d6e1, before the mixer and era commits rewrote the class literal, the load emitters, the pack fields and the pinned-pack tests: 8 files, 35 hunks); the node agent aborted cleanly and the cache agent is rebasing onto b105a55 as a squashed commit with hot: None kept; the composed class under test is unchanged by the cache code, so the igneum-pow and packfile checks, the igneumd and igneum-miner rebuild and gate run 2 proceed on b105a55 now.
21:45. PC 2 defect: since a job's /api/resume at 21:25:11Z the 0.3.9 app answered ok and never restarted the NVIDIA miner (nor the iGPU one): the 5090 worker "off" at hash 0 holding 1.7 GB, so agg-cost-pc2-2's mining phases are void (its idle phases run; closes about 21:52Z) and the devnet has been short PC 2's rate since 21:25. The 0.3.10 restart should bring the miners back; the shipper confirms PC 2's 5090 STATUS rate after the 0.3.10 line, else the aggregation-cost agent's restore script (tools/proving-v1/pc2-agg-cost-restore.ps1, 30 s) runs. Defect for the next cut: a resume that answers ok without a miner restart; the app must re-check the miner processes after a resume and report a failure. The aggregation-cost agent gets a 20-minute re-run slot on PC 2 after the 0.3.11 suites.
## 21:46 PC 1 is on 0.3.10; the era job has the go
PC 1 restarted on 0.3.10 at 21:40:41Z (engine run win-ae432dc7-20261005-214041), the 5090 at 141.4 MH/s by 21:45:17Z; "go PC 1" to the era job at 21:46 (fetch-ca2-era-20261005, zip f79c0607...; the 5090 then the AMD card; 6 to 10 min). The mixer daily-build job follows it, then the 5090 power sweep, Ember's build and fetches, the repro re-run, the AMD sweep (conditional), Ember's run. PC 2 is still on 0.3.9 at 21:45 (its restart pending); the 0.3.11 suites publish after its 0.3.10 line and a confirmed 5090 rate. Every measurement before the restart on PC 1 (hot table 21:29 to 21:35) stands: it did not straddle.
21:50. The resume fix (the miners not restarted after POST /api/resume: PC 2 since 21:25Z tonight, the Mac this afternoon) is assigned to the proving agent on 0.3.11's app branch (engine.rs resume: restart every enabled card's worker and a stale pack export, re-check within one tick, a state-machine unit test plus the known-failed case from PC 2's log); rollout plan 8a. If its commit is not in hand at the app cut, it heads 0.3.12's list.
## 21:52 gate G4 run 2 PASS on the composed class; the 0.3.11 suites are packing for PC 2
G4 run 2 (21:46:36 to 21:51:31 UTC, b105a55 = era + mixer, hot None): every check true; 181 / 124 blocks around DAA 180; v3 ids e3 2d278041ba482dba, e4 2ae786d294a8a59d, e5 bc36813df2f41b5f on all three miners; the v2 epoch-0 id 8f8806638d59850f unchanged from run 1; 0 rejected; one sink 712c1b212091dcdc at 303/303/303; the switch line on 3 of 3; cache ready 178 / 181 ms. G4 is GREEN. The 0.3.11 suite job is packing from the ca2-v3 worktree (fork 79bd8e10, main b105a55; build-inputs.zip 9,563,672 bytes sha256 bf89ab4c...) for PC 2, which is on 0.3.10 since 21:49:41Z with its 5090 worker back on the first try.
21:53. The 0.3.11 suite job is published: build-20261005-215219 to PC 2 (fork 79bd8e10, main b105a55; linux build 30 min, tests 35 min: kaspa-consensus-core, igneum-exec, kaspa-pow, kaspa-consensus, igneum-miner, kaspa-p2p-flows, igneum-app); PC 2 is on 0.3.10 with its 5090 back. The worktree freeze is lifted for the node agent (the run-2 summary commit, then the cache merge). The proving agent's resume fix waits on a test run behind the Mac's held lock slots.
21:55. The 0.3.10 rollout: both PCs mine on 0.3.10 with the rebuilt workers (PC 2 120.6 MH/s at 21:54:33Z, PC 1 141.3); the hand nodes (21:49:38Z, 21:49:50Z) and the seed (21:50:15Z) on 21d4c73c, digest 1f4b4425 everywhere; not yet on 0.3.10: the US laptop 37ba0461 (installer downloaded 21:40:52Z, app not back after 13 min; nothing to drive from here) and Sam's Mac (quit since 20:47Z). Open on PC 2: the prover fails with "CudaClientError: Connect(PermissionDenied)" since the restart (three shards 21:49:56 to 21:50:32Z); the likely cause is the sp1-gpu-server socket handling of the aggregation-cost jobs; the proving agent owns it and publishes a fix after the suite job (PC 2 is the suite job's until it closes). The ca2-v3 tip is 63dabb2 (run-2 summary and doc, the G6 job id recorded).
## 21:56 gate G6: the first PC 2 job failed on the known kaspa-consensus flake; split re-run
build-20261005-215219 (21:53:01 to 21:55:48Z): Linux build ok (igneumd 49,164,264 bytes sha256 11979b49..., igneum-miner d25a8270..., igneum-app 68007173...), igneum-app tests 78 + 26 + 8 passed; the node stage exit 101: kaspa-consensus 96 passed, 1 failed, processes::finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list, UnexpectedDifficulty(..., 487112384, 487129578) in mine_on_all. This is the flake the 0.3.10 cut met on the same base 21d4c73c under the six-package parallel run (release-0.3.10.md: it passed alone twice, job build-20261005-182804, 97 passed), a timing-dependent difficulty in the test helper, not a v3 change (v3 touches no finality code). Per the gate rule the publish stops here until the suites are green: job 2 of 3 (kaspa-consensus alone) is published now; job 3 of 3 (the other five crates) follows; the flake itself goes on the next-cut list (make mine_on_all deterministic under parallel load).
The proving v1 app branch is final at a223ca9 (6dc686a plus the resume fix with the PC 2 case as a unit test); rollout plan 8a updated. PC 2's prover fault is the root-socket class (agg-cost-pc2-1 ran the host as root in WSL2 and left /tmp/sp1-cuda-0.sock owned by root; the app's prover has failed every shard since 21:25:24Z); the proving agent's pc2-socket-fix.ps1 (60 s, miners untouched) runs between my two suite jobs.
## 21:57 the era PC job is done; a 2.2x CPU-verifier regression on ca2-v3 HEAD (gate item); the PC 2 order
Era job run-ca2-era-pc1-20261005 (exit 0, 301 s), both cards restored, the 9070 XT ON the bus at run time (gfx1201, 32 CUs): first rows 18.96 to 19.21 MH/s on the era packs on the 9070 XT, self-test PASS, the 2^24 fingerprints equal to the Mac's (era-4 3ace11ad84c053ae, era-5 a8897d82adceb4a1); the full 5090 and 9070 XT tables with the spread per card follow. "go PC 1" to the mixer daily-build job at 21:57.
GATE ITEM (found by the era agent, two binaries on the same v2 input, checksum 19297e99c7b9a55e, same minute, load 4 to 5): the CPU verifier on ca2-v3 HEAD (88dafbc) takes 1.332 ms per warp and on the era branch 1.310, against readwidth's binary at 0.604 and 0.606: the mixer branch's derive_items / Shape path costs 2.2x at m = 1, on the v2 path the live devnet verifies with. The mixer agent's "loaded box" reading of its 1.31 to 1.36 ms v2 figure was the code, not the load. Ordered: find and fix on ca2-mixer, restore v2 to within 5% of 0.604 ms measured the same way, re-measure v2 / x4 / x8 on the fixed binary (the C19 figures too), then the node agent merges it; the publish waits on it (G3's suite does not catch a slowdown; the 10 ms gate and the pool and IBD figures depend on it). Next-cut rule: a verifier benchmark with a pinned bound in the crate's CI.
PC 2 order: suite job 2 of 3 (kaspa-consensus alone, running), the proving agent's socket fix (60 s; the aggregation-cost restore is skipped), suite job 3 of 3 (the five other crates), the aggregation-cost re-run (20 min), the prover-floor agent's windows (through the proving agent). The proving app branch: a223ca9 is the last code change (8b47073 docs only after it).
21:58. ca2-cache is rebased: one squashed commit 1950661 on ca2-v3 63dabb2 (fast-forwardable; history under tag ca2-cache-history-2026-10-05); V3_CLASS = { era: None, hot: None, ..MX4 }; every hunk kept the ca2-v3 side and appended the hot code; 53 + 19 tests; the pinned v2, mx4, era and readwidth packs untouched; the eight hot packs re-exported with unchanged vectors, 96/96 and the pre-rebase fingerprints on Metal and Apple OpenCL. The node agent merges it after the mixer's verifier fix.
## 21:58 layers 4 and 8 decided IN on the PC rows (ca2-era 78c0ee4)
| Card | v2 MH/s | six eras min / median / max | Spread | Fingerprints = Mac | Latency-bound share |
|---|---|---|---|---|---|
| RTX 5090 (PC 1, CUDA, 1 warp per block) | 137.2 | 136.18 / 136.44 / 138.01 | 1.3% | 7/7 | 1.01 |
| RX 9070 XT (PC 1, OpenCL, group 256) | 18.09 | 18.61 / 18.93 / 19.21 | 3.2% | 7/7 | 0.95 |
| M5 Max (Metal, 5 x 2^24) | 27.68 | 28.35 / 28.48 / 28.58 | 0.8% | 7/7 | 1.06 |
All under the 5% rule; the CPU verifier 1.00 to 1.02 of v2 within one binary; the dataset build with the scatter store 20.7 to 21.8 ms against 21.1 linear on the M5 Max. Chip line: 512 B per hash, 120 to 128 distinct 64-B lines, the SRAM mirror is the whole dataset every hour (a windows-union census over 300 programs); the interleave buys nothing against a chip with a programmable address decoder (stated in the doc); the stride is a bijection with no cryptanalysis yet. Commits: b105a55 (implementation, history under tag ca2-era-pre-squash), c570da3, 669a27a, 78c0ee4 (the PC rows). The suite job 2 of 3 is build-20261005-215712 on PC 2 (kaspa-consensus alone).
21:58. ca2-v3 tip fbf958e: ca2-cache 1950661 fast-forwarded (no conflict), V3_CLASS = { era: None, hot: None, ..MX4 }; igneum-pow 53 + 19 tests, packfile-test 0 failures; the fork's check against the new crate running; the fork stays 79bd8e10. The node doc carries the CPU hash rate across the switch in run 2 (about 30% under v2 on the CPU interpreter, approximate, a shared Mac) and the verifier before/after slot for the mixer fix. Gate run 3 on the fixed tree follows the mixer fix.
## 21:59 0.3.10 is shipped and merged to master (cde561c, pushed 21:58:17Z)
The shipper's report: master cde561c (the 0.3.10 merge) + 1f0d62c (the plan); fork release-0.3.10 21d4c73c; digest 1f4b4425... on every node; six suites green on 21d4c73c on PC 2 with the same ban_is_decided flake under the parallel run (passes alone twice), recorded for the c4 agent. Open from it: PC 37ba0461 (the US laptop) stuck in its install since 21:41:16Z; Sam's Mac quit since 20:47Z; PC 2's prover dark (the root-socket cause is now named, the fix queued); C1 at 16:00Z; the /api/resume no-op (fixed on the 0.3.11 app branch).
Consequence for 0.3.11: the main tree base is now master cde561c, so the integration merge is ca2-v3 (8ea6740 plus the mixer fix) and ca2-coord into master, with the app branch a223ca9 (on 5b0d54f, which master contains). The fork base stays 21d4c73c (= release-0.3.10's tip), so the fork merge is clean by construction.
## 22:00 G6 job 2 of 3 green; the merge into master dry-run
build-20261005-215712 (21:57:12 to 21:59:45Z): kaspa-consensus alone 97 passed, 0 failed, 3 ignored (ban_is_decided ... ok), every stage ok. The PC 2 prover socket fix runs now (the proving agent, 60 s), then job 3 of 3 (consensus-core, igneum-exec, kaspa-pow, igneum-miner, kaspa-p2p-flows and the app tests).
Merge dry-run into master cde561c (a scratch worktree, aborted): ca2-v3 (8ea6740) conflicts in docs/bench-log.md and proto-opencl/host.c; ca2-coord conflicts in docs/bench-log.md, proto-cuda/nvrtc/packfile.h and proto-opencl/host.c (master's 0.3.10 merge brought pack-loop's packfile.h and opencl-rdna4's host.c). The bench log is append-only (keep both); packfile.h and host.c take the ca2-v3 side (it carries the pack-loop rule plus the class, era, mixer and hot fields) re-checked against master's hunks. The integration merge is the ship's first step.
## 22:01 the verifier regression bisected to the mixer's 0fc0ad1; the mixer PC 1 job is running; PC 2 queue
Bisection (mixer agent, measure lock, 21:59 UTC, same input 19297e99c7b9a55e, avg of 50, two rounds): readwidth e752fc7 0.607 / 0.609 ms per warp; the ca2-v3 seam 6c75dad 0.610 / 0.609; the mixer's 0fc0ad1 1.332 / 1.316; ca2-v3 HEAD 88dafbc 1.325 / 1.347. The 2.2x is in 0fc0ad1's derive_items / Cache path at m = 1 (not the era layout, not the load). Three candidate fixes building (the constant line mask back in Cache::line; an m == 1 fast path that is readwidth's loop verbatim; both); the one that restores 0.61 goes on top of ca2-v3 8ea6740 with the six mixer commits rebased, measured the same way; then the node agent merges, rebuilds and runs gate 3.
PC 1: fetch-mixer-x4-20261005 and run-mixer-x4-pc1-20261005 published 21:59:32Z (zip 152fcf93...; workers e6007918... and e4334aaf... from ca2-mixer; the five packs; 25-minute timeout). PC 2: the proving agent's socket fix (60 s) now, then suite job 3 of 3, then the prover-floor agent's toolchain check (3 min) and its 60-minute niced sp1-gpu-server build (CPU only; the shipped server panics on any card under 24 GB, sp1-gpu builder.rs 35 to 39, and allocates every prover at Setup: the 13.9 GB floor's cause), then the aggregation-cost re-run (20 min), then the prover-floor measurements.
0.3.11 ship template (release-0.3.10.md section 7): push the release branch, `gh workflow run windows.yml --ref <branch>`, `node tools/ship-app.mjs 0.3.11 --node <fork worktree> --branch <branch> --public --activation-height N4 --deadline-note "program class v3 + proving v1" --notes "..." [--from ci]` with the override object carrying every switch; gh must be on igneum-labs; the pre-push hook flips two site files (restore with `git checkout -- site/`).
## 22:03 PC 2's prover is back; suite job 3 of 3 published
socketfix-pc2-pv1 (22:01:14 to 22:02:12Z): /tmp/sp1-cuda-0.sock owned by root removed (the aggregation-cost job's run), the prover switched off and on; the next shard (block 89011 shard 0) "proven and submitted in 34 s" at 22:02:13Z and paid 0.93116546 IGN at 22:02:24Z. PC 2's prover had been dark from 21:25:24Z to 22:02 (the root-socket class; the CI check tools/ci/prover-socket-check.sh now fails any playbook without the two restore lines). Suite job 3 of 3 (consensus-core, igneum-exec, kaspa-pow, igneum-miner, kaspa-p2p-flows and the app tests; main 8ea6740, fork 79bd8e10) is packing and publishing from the ca2-v3 worktree now. After it on PC 2: the prover-floor agent's toolchain check and its 60-minute build, then the aggregation-cost re-run.
## 22:06 decided: mixer x8 into v3; the verifier fix found; ca2-v3 at 795472e
The mixer PC 1 job (22:00:08 to 22:04:39Z, both cards restored, the 9070 XT present): the daily 1 GiB build per pack, two dispatches, wall ms: RTX 5090 v2 25 / 23, mx4 24 / 24 and 23 / 25, mx8 23 / 23 and 23 / 23 (cache 4); RX 9070 XT v2 74 / 74, mx4 77 / 75 and 73 / 74, mx8 72 / 76 and 75 / 74 (cache 8 to 9). The build is latency-bound on every card; the x8 rule's build half passes with 13x to 40x margin; its verifier half passed at 2.79 ms per warp on the loaded core. DECIDED (delegated): x8 into v3; V3_CLASS = { era: None, hot: None, ..MX8 }; the chip row at x8 reads 0.92x with the 3x factor (the claim "under 1x with the factor", margin stated). Every mixer pack's fingerprint equals the Mac's on both cards (v2 25f96e7dce90bd4e; mx4 6f48d5a2aa0dbe5f, 73caaebb28e808fe; mx8 7c28cfb06c5c65a9, bbb183f72692f840); hash rates at the v2 rate on both (5090 136.5 to 137.4, 9070 XT 18.0 to 18.2 at every class).
The verifier regression is found: not the mask but inlining; the item loop inlined into MemhardCpu::fetch runs at 1.33 ms per unit, the same loop out of line (#[inline(never)], one instance per cache size, the line mask a constant) at 0.60 to 0.62 against readwidth's 0.60 to 0.64 in the same minute. The fix, the MX8 V3_CLASS, the pinned pack re-export and the final v2 / x4 / x8 session land as one commit on ca2-v3 795472e (the node agent merged the mixer's 16dfd1e as 4e733bb with two one-line field fixes; 53 + 4 + 19 + 7 tests, packfile 0 failures, the fork check clean). Then: the node agent merges, rebuilds, gate run 3 on the final class; the six era packs and the pinned v3 pack re-exported on it; the final bit-exactness and G2 (1,000 random hashes per card re-hashed by the CPU) job on PC 1.
## 22:08 G6 job 3 failed on a stale fork test (era inside the class); job 4 on the final tree
build-20261005-220351 (22:04 to 22:06:55Z): igneum-exec 17, igneum-miner 18, consensus-core and p2p-flows green, the app 78 + 26 + 8; kaspa-pow 13 passed, 1 failed: igneum::tests::program_class_v3_seeds_hash_their_own_program_over_their_own_cache asserts the program's class equals V3_CLASS with era: None, but the merged crate carries the drawn era inside the class (left: era Some(EraParams { ... stride_mul 3969900165 ... }); right: era None). A stale fork test written before the era merge, not a behaviour fault; the node agent fixes it on the fork. Correction to the G6 note: the PC's test stage DOES run the v3 engine test, so the PC job is the evidence. Job 4 (the five crates and the app) runs on the final tree once the fork fix and the mixer's verifier fix (with V3_CLASS = MX8) are merged; the publish waits on it. PC 1: the 5090 power sweep has the go (8 min), the AMD sweep may follow on its own presence probe; the final v3 vectors and G2 job (the era agent, 1,000 hashes per card re-hashed on the Mac) is being prepared for about 22:40.
22:09. Fork ca2-v3-node 89dfcb95: the v3 engine test asserts what it meant on the era crate (LoadClass { era: None, ..class } == V3_CLASS and class.era.is_some(); the era bytes equal the seeds'; another era seed keeps the program id, draws another era class, hashes another pow, adds no day cache); Mac `cargo test -p kaspa-pow --features igneum-pow` 14 passed. Main-repo tip 6a705a2. Waiting on the mixer's fix commit for the merge, the rebuild, gate run 3 and suite job 4. PC 2: the prover-floor agent's 3-minute toolchain check has the go; its build waits for job 4.
22:11. PC 2: the prover-floor toolchain check done (floor-toolchain-1, 22:10:03 to 22:10:06Z: nvcc 12.8, cmake 3.28.3, gcc 13.3, clang 18, cargo 1.99.0; no go, so the rebuilt server drops native-gnark, the Groth16 wrap that compressed proofs never use; 16 cores, 30 GB WSL RAM; the live server untouched); its 90-minute build (sm_86, sm_89, sm_120 after C26) is HELD behind suite job 4 (the gate), with a 22:40 fallback: if job 4 is not published by then, the build goes first. Consequences C26 (the arch list, the card, the packaging row, the verify-segment run, "24 GB" kept until the 3060 proves) is with the prover-floor agent; C27 (publish-jobs.sh add runs the prover-socket and bash-body checks and refuses on failure) is being wired by the reviewer's sub-agent, no collision.
## 22:11 the verifier fix and the x8 class are committed (ca2-mixer 1ab8b21); the final code is in
Before / after, the era agent's way (one measure session, 22:07 UTC, readwidth e752fc7's binary beside the fixed one, the same v2 input, load 5.5): readwidth 0.607 / 0.610 ms per unit; the fixed binary 0.609 / 0.611 (was 1.332 / 1.316 on 0fc0ad1). On the fixed binary: x4 1.238 / 1.237 (2.0x v2, worst cold 1.40), x8 2.077 / 2.058 (3.4x, worst cold 2.15), x8 on the devnet seeds 2.058; so the x8 class verifies at 2.1 ms per warp on a loaded core, 4.8x inside the 10 ms gate. Cause and fix: the item loop inlined into MemhardCpu::fetch ran at 2.2x whatever the mask; derive_items_mask #[inline(never)], one instance per cache size (2^26 to 2^30) with the line mask a constant, restores 0.61. V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }; mx8-genesis (program id e323b9dcaf283a6f, fingerprint 7c28cfb06c5c65a9) and mx8-devnet-epoch0 through the chain path with the era inside (class mx8-erad810f22d, program id 73bcbfe8ccf988f1, unit 0 lane 0 d424577fce4a7a60, fingerprint 90f794dd556f7a3b on Metal and Apple OpenCL, 3/3 + 3/3, 96 of 96); the suite 53 + 4 + 19 + 7. chip-model-v3.md headline: x8, 0.92x with the factor, margin 8% on the factor and 9% on the budget; x4 kept as the measured candidate. Owed: the composed mx8-devnet-epoch0's fingerprint on the PCs (in the final-vectors job). Next: the node agent fast-forwards ca2-v3, re-checks, rebuilds, gate run 3; suite job 4 publishes when the tree is settled; the era agent builds the final-vectors and G2 package on 1ab8b21.
22:13. Suite job 4 published: build-20261005-221237 (main d233fa1 = ca2-v3 with the mixer fix merged, V3_CLASS = { era: None, hot: None, ..MX8 }; fork 89dfcb95 with the engine test fixed; the five crates and the app). The node agent's checks, rebuild and gate run 3 on d233fa1 follow; the era agent's final-vectors and G2 package is being built on the final class; PC 1 is the AMD sweep's until about 22:25.
22:15. The era final-class package is ready (zip igneum-ca2-era-pc1b.zip sha256 cb0e9db07e304b11fd4c0591351af46090442ea4f51d60eb86945b96bd28aba3; workers f8d19f0a... and 87647c15... from 1ab8b21 plus the era branch; seven packs on the final class: mx8-devnet-epoch0 and era-0 to era-5, program id 73bcbfe8ccf988f1; Mac 7/7 on Metal and Apple OpenCL, fingerprints equal: mx8-devnet-epoch0 a6752e037514c91a, era-0 64c0ee90bac42624, ..., era-5 43673acc89954d5e). G2 method: one serve-mode job of 1,024 nonces at target ff..ff per card for era-0 and the pinned pack, every nonce a "found g2" line, re-hashed on the Mac with igneum-pow hash-bound --count 1024 (dry run through Apple OpenCL: 1,024 of 1,024 on both packs). Its PC 1 go follows the 5090 power sweep (ahead of the AMD sweep). C24 and C27 closed on branch bash-body-check (7adb1ca, 6805125); the integration merge takes its prover-socket-check.sh over proving-v1's and puts pc1-cpu-prove.ps1 on the allow list.
## 22:16 the 5090 power-limit sweep (PC 1, relay #224, 22:09 to 22:15:37Z); the G2 job has the go
| Cap | Limit W | Draw W | MH/s | MH/W | SM MHz | Memory MHz | Busy |
|---|---|---|---|---|---|---|---|
| 100% | 575 | 316.2 | 115.42 | 0.365 | 3,051 | 13,801 | 92.9% |
| 80% | 460 | 316.1 | 115.60 | 0.366 | 3,050 | 13,801 | 92.5% |
| 65% | 400 (the floor) | 310.6 | 114.46 | 0.369 | 3,051 | 13,801 | 90.1% |
| 50% | 400 (clamped) | 302.4 | 109.20 | 0.361 | 3,050 | 13,801 | 87.5% |
The card draws 302 to 316 W under this program whatever the cap, so a cap above 400 W never binds; the readwidth, era, hot-table and mixer numbers taken at 431 W sit on the flat part of the curve (within 1% of stock); best per watt 65% (400 W) at 0.369 MH/W, a 0.8% hash cost. The cap was restored to 431 W and read back. (The app's 0.3.9 rate of 124 to 141 MH/s in the STATUS lines against 115 here: the API's hash_now sampled every 5 s under the sweep's own load; the bench rows of 136 to 137 MH/s are device time.) "go PC 1" given to the era agent's final-vectors and G2 job at 22:16; the AMD sweep follows it on a fresh probe.
## 22:16 gate G6 GREEN on the final tree
build-20261005-221237 (main d233fa1, fork 89dfcb95, 185 s): every stage ok; kaspa-consensus-core 108 (2 ignored), igneum-exec 17, kaspa-pow 33 + 14 (the v3 engine test, with the igneum-pow feature), igneum-miner 18, kaspa-p2p-flows 7, igneum-app 78 + 26 + 8; 0 failed; with job 2's kaspa-consensus 97 alone, G6 is green. PC 2 goes to the prover-floor agent's 90-minute build (its go at 22:17), then the aggregation-cost re-run, then the prover-floor sweep. Gates: G3 green (the Mac suites on the final class: 53 + 4 + 19 + 7 crate tests, the Metal fuzz, edge, stats and determinism runs, the scratch tests), G4 green (runs 1 and 2; run 3 on the final x8 + era class pending), G6 green; G1 and G2 pending the era agent's PC 1 job (running from 22:16); G5 (the Windows and Mac workers from the same commit) is the ship's build step on the merged tree.
## 22:18 the spec and the public copy carry the final class
docs/spec/01-lottery-hash.md on ca2-coord: 1.8.5 (the mixer x8 form with the measured costs), 1.13.1 (the era draw: stride, interleave, windows, the devnet stand-in, the measured spread), 1.17 (the class v3 vectors), 1.5 (the cache note); earlier tonight 1.12 and 1.13.1 (epoch_len), 1.13.2 (R1 and the emulation rule), 1.13.3 (option C and the step mapping), 1.4.5 and 1.4.6 (generator 3, the class and era in the pack), and 4.3. Public copy: level 1 on the hero and the abstract ("Built for graphics cards. A custom chip gains under 2x, and the model and the bounty are public."), the limits bullet rewritten on the chip row (0.92x with the allowance, approximate; the margin on the numbers page; the next lever named), the level 3 table filled from the final numbers in counter-asic-2-public.md (the bench page section is written from it at the ship). Waiting: the era agent's PC 1 job (G1 on the final class and G2), the node agent's gate run 3; then the integration merge and the ship.
22:19. docs/analysis/proving-methods.md (branch proving-methods e7e0db7, not consensus): top recommendation re-size SP1's own GPU server (the floor is its code: the under-20 GB panic at sp1-gpu builder.rs:37, trace buffers at the maximum shard, a CUDA mempool that never releases), S_p as the dial, one server per card on rigs; the pinned ids stay; fallback and the Apple route: RISC Zero as proof-system version 2 behind the ProofSystem seam (8 GB at po2 19, 16 GB at po2 20, shipped Metal; 3 to 4 agent days); no 12 GB card has run a prover here, so a 4070 or 3060 in the loop is the first action. That branch merges into the 0.3.11 main tree as documentation (no code). evidence.md rows 17, 18 and 19 are rewritten on the measured class v3 (ba8379d).
22:20. The numbers page's Counter ASIC section is written as the bench-log entry "Counter ASIC 2.0, the numbers" (the page is built from the bench log; the litepaper links /bench#counter-asic-2-0-the-numbers); the litepaper's verifier figures moved to the v3 class (2.1 ms per warp on a loaded core, 4.8x inside the gate; the cache 512 MB from year 4). Everything public now carries the final class except the PC rows of the final-class packs and the G2 counts, which the running PC 1 job supplies.
## 22:20 gate G4 run 3 PASS on the final class; the node side is final
Fast-time 3-node network on ca2-v3 d233fa1 (x8 + era + the verifier fix) and fork 89dfcb95, 22:15:04 to 22:19:44Z: every check true; 182 / 122 blocks around the boundary; the v3 ids agree on all three miners; 0 rejected; one sink at 303/303/303; the switch line on 3 of 3; cache ready 179 / 191 ms. Final tips: ca2-v3 5eb2331 (docs only since d233fa1), ca2-v3-node 89dfcb95, both clean; the gate network stopped. Gates: G3 green, G4 green, G6 green; G1 and G2 on the final-class PC 1 job (running since 22:16); G5 at the ship's build step.
## 22:21 gate G4b added: the Mac must mine v3 (the app's --prepare-packs gap)
The node agent's owed item is an app gap, confirmed in the 0.3.11 app branch: engine.rs miner_args pushes --prepare-packs only for non-Metal workers (line 1468 on a223ca9), so the Mac app's Metal worker would get no prepare pack and under class v3 would answer `need` at the first v3 epoch (every Mac stops: the 18:23Z class). Closes in hand: (a) the proving agent adds the flag for every worker (packs/prepare on macOS) with a unit test on miner_args for a Metal card, on the app branch above a223ca9; (b) the node agent runs gate 4: a real Metal miner on this Mac across a v3 boundary on the fast-time network through the miner's --prepare-packs flow. The ship does not go without (a) in the app tree and (b) green; recorded as G4b in the rollout plan's gate list.
22:23. G4b (a) done: app proving-v1 e0de2ab on 5b0d54f: miner_args pushes --prepare-packs for every worker through prepare_packs_arg() (packs\prepare on Windows, packs/prepare elsewhere), the OpenCL-only --job-nonces kept, and start_miner now gives the Metal worker the app data folder as cwd (it had None, which would have broken the relative path: a second latent Mac fault closed); unit test every_worker_gets_the_prepare_directory_in_the_platform_form; 113 + 27 + 8 passed. The 0.3.11 app tree's final code commit is e0de2ab. (b), the Metal gate run, is with the node agent.
22:23. PC 2: floor-build-1 (22:17 to 22:21:12Z) FAILED at cargo exit 101 after 240 s with the error not uploaded (the playbook named the cargo log with a timestamp and could not find it; fixed: a fixed path, the error lines printed on failure); floor-build-2 (the same 90-minute shape) has the go at 22:24; the aggregation-cost re-run follows it. PC 1: the era agent's final-class job runs (from 22:16); the AMD sweep follows on a fresh probe.
## 22:24 gates G1 and G2 GREEN on the final class
Job run-ca2-era-pc1b-20261005 (22:16 to 22:21Z, 304 s, both cards restored, app 0.3.10, the 9070 XT present): the seven final-class packs' 2^24 fingerprints equal on the 5090, the 9070 XT and the M5 Max (mx8-devnet-epoch0 90f794dd556f7a3b; era-0 8e8e070db4eea52d, era-1 891c01b8563bb47e, era-2 e54279fed2831b5d, era-3 77e0ba8abbd0ae62, era-4 d898d8f4f2e7684b, era-5 a6927db380f7efb2); G2: 1,024 of 1,024 per card on era-0 and the pinned pack re-hashed by the Rust verifier. Rates on the final class (MH/s): 5090 135.90 to 137.70 (spread 1.3%), 9070 XT 18.59 to 19.18 (3.1%), M5 Max 27.85 to 27.98 (0.5%); the pinned pack 136.10 / 18.90 / 27.80. Spec 1.17 carries these fingerprints. Gates now: G1, G2, G3, G4, G6 green; G4b (a) done, (b) the Metal gate run pending; G5 at the ship's build. PC 1 goes to the AMD sweep on its presence probe.
22:25. Consequences C29 and D11 applied. C29: the litepaper's verifier line now reads 2.1 ms per warp for class v3 on one M5 Max core at load average 5.5 (the fixed crate; worst cold 2.15), 3.4x the v2 verifier's 0.61 ms, the gate leaving 4.8x (4.6x on the worst cold unit); the "about 1.3 ms quiet" scaling is struck: it came from the slow binary's 2.79 ms session (21:40), and the fixed binary's session (22:07) matched readwidth's quiet v2 figure within 1%, so the fixed numbers are near-quiet. D11: "and the bounty" struck from the hero and the abstract, "standing" from level 1; the copy says "a bounty follows the external review"; the bounty is named only once escrowed (funding.md rule 3; USD 50,000 not funded); the project lead decides (D11). The level 3 table and the numbers-page entry now carry the final-class rates (5090 135.90 to 137.70, 9070 XT 18.59 to 19.18, M5 Max 27.85 to 27.98).
## 22:25 the fourth eGPU drop; the era branch final; PC 1 to Ember
The 9070 XT is absent again at 22:24:09Z (relay probe #230: no Sonnet or USB4 router device), the fourth drop today, after serving the hot-table, era (twice) and mixer jobs between 21:31 and 22:21; the AMD clock and power sweep did not start and its rows are owed with this time and reason; the hardware-events table in the rollout plan carries the flapping for the morning. ca2-era is final at 95955c3 on ca2-v3 5eb2331 (six commits: the implementation, the Mac rows, the PC round 1 rows, the G2 hash-bound --count flag, the final-class re-export, the round 2 rows; 53 + 30 tests). "go PC 1 build" given to Ember Tune (its two fetches and the 8-minute Windows app build), then its 8-minute run on the 5090 baseline. PC 2: floor-build-2 running (to about 23:50), then the aggregation-cost re-run, then the prover-floor sweep.
## 22:27 the integration merge, in the node agent's hands; the ship order
A dry run of ca2-v3 into master in a scratch worktree conflicts in docs/bench-log.md (append-only, both kept) and proto-opencl/host.c, where taking ca2-v3's side would drop master's 0.3.10 hunks (the PCI topology and duplicate-platform detection, the select read-back): it must be resolved by hand keeping both. ca2-v3 also lacks readwidth's last three commits (d0018cf the OpenCL scratch local-buffer rule, e752fc7 read-width.md and the overflow fix, 30ff674 the per-watt rows). So the node agent, as ca2-v3's owner, merges readwidth 30ff674 and origin/master (cde561c, 1f0d62c) into ca2-v3 beside gate run 4, re-runs the igneum-pow suite, the packfile test, test-generic.sh on Apple OpenCL and the fork check, and sends the tip. ca2-coord (docs, spec, site, evidence, the plans) stays on its a9e002c base: its code files are untouched readwidth copies, so its merge onto that tip takes ca2-v3's code and brings only the documents (a rebase attempt replayed readwidth's own commits and was aborted). Ship order: master <- ca2-v3' <- ca2-coord <- the tooling and analysis branches (consequences, bash-body-check, amd-prove, card-lifetime, proving-methods, asic-history when it lands), the app branch proving-v1 e0de2ab (on 5b0d54f, in master), the fork ca2-v3-node 89dfcb95 (on 21d4c73c, release-0.3.10's tip). The ASIC-history agent (a202a09dcd24ba1d3) resumed at 22:52 (its clock) in ../igneum-wt-asic-history; its ranked additions are Counter ASIC 3.0's input after the publish.
22:30. C31 applied (a7be43f): the litepaper's "12 GB or more proves full shards" struck (one page, one number: 24 GB); the dataset's step schedule written as the gate 1 proposal on the site and litepaper (D4 is the project lead's); the merge rule "take proving-v1's evidence row 16 and its proving sentences" in the rollout plan. PC 1: Ember's build done (build-20261005-222558, 97 s, 6 outputs verified), its tune run on the 5090 has the go. PC 2: floor-build-3 (with the pinned Go toolchain; builds 1 and 2 failed on a missing go, the first unreported by a log-path bug) runs, 25 to 45 minutes expected.
22:32. C32: an app update clears the jobs folder (the 0.3.10 install took PC 1's AMD kit, fetched at 21:23:59Z; the 21:05 reading "cleared by fetch jobs" was wrong), so two rules enter the rollout plan (4a): the 0.3.11 update-now reaches PC 2 only after floor-build-3 closes (PC 2 last on the machine list), and every fetch-then-run pair re-fetches after an update, with a kit presence check at the top of every run playbook.
## 22:33 gate G4b GREEN: the Mac mines v3 end to end; a second Metal outage found and fixed
Gate 4 (22:25 to 22:31Z, a real Metal miner through the miner's --prepare-packs flow on the fast-time network, igneum-bench from ca2-v3 00c55aa): three v3 prepares with the pack dir and the class and era tokens; the worker's v3 prepared lines (252.5 ms the first: program 55.5, dataset 196.9, cache fill 0.9, build 41.1; then 51 and 46 ms with the day resident); every swap "with no pause"; 124 blocks accepted on v3, cpu re-check mismatched 0, need 0, no mismatch or refusal; chain 182 / 123 across the boundary, 0 rejected, one sink. Found and fixed before the run: the Metal worker's serveDataset built every day with the Swift version 2 construction keyed by day only, so a v3 program would have hashed over an x1 dataset and every found would have been refused by the CPU re-check, the same fleet-outage class as the app's missing flag; now a v3 prepare builds the day from the pack's memhard.metal and the store keys datasets by (day, class, era) (00c55aa). Two Mac outages caught by one gate run that the CPU-miner gates could not see; the rule for the next cut: every worker path (Metal, CUDA, OpenCL) mines across a boundary in the gate network before a class change ships. The integration merges (readwidth 30ff674, then origin/master) are in progress on ca2-v3 with the conflicts resolved by hand keeping both sides.
Gates: G1, G2, G3, G4, G4b, G6 GREEN; G5 at the ship's build step. The ship waits only on the merged tip and its checks.
22:33. H = 210,000 lands about 18:45Z on 6 October (DAA 137,041 at 22:32:54Z, 1.002 DAA/s since 15:40Z); the 16:00Z check keeps 2.75 hours. PC 2: floor-build-3 green at 22:32:29Z (240 s on the warm target; the patched sp1-gpu-server 166,768,224 bytes sha256 5568108b..., v6.8.1 c84ada1e with patch 700173fe, sm_86 sm_89 sm_120, the Go tarball's sha256 matched; the live server untouched, miners never stopped); "go PC 2 sweep" given for floor-sweep-1 (9 points, 8 to 10 min) ahead of the aggregation-cost re-run.
## 22:38 the merged tip is in; the 0.3.11 ship is assigned
ca2-v3 fa3c932 (code 49c7e78): readwidth 30ff674 merged at 3566afd (emit.rs two hunks, HEAD's superset; packbench.swift's footprint lines added to the hot-aware RESULT line; bench-log both; read-width.md readwidth's version), origin/master 1f0d62c merged at 49c7e78 (host.c one struct hunk, both fields kept; master's topology, duplicate-platform and select read-back code and ca2-v3's class, era, mh_word, hot and mixer code all auto-merged; packfile.h no conflict); checks all green on 49c7e78: igneum-pow 53 + 4 + 19 + 7, packfile-test 0 failures, the NVRTC emulator test PASS (17 sampled hashes equal to hash-bound), test-generic.sh on Apple OpenCL PASS, the fork's check clean. Fork ca2-v3-node 89dfcb95.
The ship is assigned to the 0.3.10 shipper (ae892a8b0f78fe31c) with every input (rollout plan sections 3, 4, 4a, 7, 8, 8a): release-0.3.11 from origin/master; merges in order ca2-v3 fa3c932, ca2-coord, the app branch proving-v1 (e0de2ab plus the prover log-line commit), bash-body-check e3bd761, consequences, proving-methods e7e0db7, asic-history if it lands; the fork 89dfcb95 under the release tree's vendor/; the override object with every switch; N4 = (DAA at publish + 14,400) rounded up to a multiple of 3,600, N5 = DAA + 14,400; the expected digest c562d70e...; the deadline note "program class v3 + proving v1"; update-now machine by machine with PC 2 last after the prover-floor sweep; the hand nodes, the seed, the digest sweep, the HiveOS package, the merge to master and the push. The node and proving agents stand by for fixes. PC 1: Ember's tune run. PC 2: floor-sweep-1 (from 22:34:11Z), then the aggregation-cost re-run.
## 22:41 the release tree is assembled; counter-asic-3.md written; PC 2 to the aggregation-cost re-run
Release worktree /Users/joshm/Projects/igneum-wt-ship0311, branch release-0.3.11 from origin/master b38f3de: merges in order ca2-v3 fa3c932 (clean), ca2-coord (bench-log both kept), the app branch 22c2363 (bench-log both; docs/evidence.md rows 15 and 16 from proving-v1, 17 to 22 from ca2-coord; the litepaper's schedule sentence from ca2-coord with proving-v1's 24 GB sentence appended), ca2-coord 57844e9 (the nine-field override line), bash-body-check e3bd761 (ci.yml both steps kept, prover-socket-check.sh from bash-body-check), consequences 99fd988, proving-methods e7e0db7, asic-history 9e4af7f; tip b968ee0; the fork worktree vendor/igneum-node-0311 at 89dfcb95 under the release tree. The checks (igneum-pow suite, the app suite, test-generic.sh on Apple OpenCL) run now; then the handoff to the shipper (the coordinator has told it to ship). docs/plans/counter-asic-3.md is written from the ASIC-history agent's seven ranked additions (the partial-store chip and the time-memory curve first, the random daily derivation as a reserve, the mixer cryptanalysis as a genesis gate, the detector and the issuance-triggered bounty, the FPGA lane, the reserve order, the vendor-share metric) with the decisions for the project lead.
PC 2: floor-sweep-1 (22:34 to 22:37:55Z, every proof verified by the unpatched host): the shard term is gone but a second floor binds at 12.7 GB measured (10.95 GB the server's own): at Setup the server pre-builds five recursion keys at the fixed 2^27 capacity (3.75 GB) plus the shrink and core keys, 9.7 GB before the first shard; a second patch (every trace buffer sized to its program or shard) follows in about 30 minutes; nothing is under 11.0 GB yet; the public line stays 24 GB. "go PC 2" to the aggregation-cost re-run (20 min); the floor's second build after it; the 0.3.11 update-now reaches PC 2 after both.
## 22:42 the release tree is handed to the shipper; the ship runs
Checks on the merge tip b968ee0 on the Mac: igneum-pow 53 + 4 + 19 + 7, the app 113 + 27 + 8, 0 failed (the OpenCL generic test wants the NVRTC emulator run first; both ran green on the same code at 49c7e78). The coordinator told the shipper directly to ship; it has the tree (its version bump 21173c4 sits on b968ee0 in /Users/joshm/Projects/igneum-wt-ship0311) and I touch nothing in it from here. Handoff message sent with the tip, the fork worktree (vendor/igneum-node-0311 at 89dfcb95), the nine-field override line, the two digests and the machine order (PC 2 last, after the aggregation-cost re-run closing about 23:02 and the prover-floor agent's build 4 and sweep 2 on the v2 patch eda49ab: every trace buffer sized to its padded need, the recursion keys 0.75 to about 0.21 GB each). The shipper reports to the coordinator and me at each step; the node and proving agents stand by.
22:43. PC 1: Ember's tune run (ember-tune-pc1-1) failed at 22:31:06Z after 47 s (the second engine exited at once; no knob touched: the 5090 at its 450 W limit, 2,850 MHz; the 9070 XT present on bus 98 at factory); a collect job reads the exit reason; Ember's re-run comes after PC 1's 0.3.11 update with its fetches republished (C32). Finding from its probe: the AMD helper's gmax range is an offset (-500 to 1000), not MHz; the probe is fixed to keep the clock knob closed on offset ranges and bound the power ladder by plimit_range. The window before PC 1's update goes to the AMD sweep (a fresh kit fetch, the two-read probe, 26 minutes if the card is present); the shipper holds PC 1's update-now until "PC 1 clear" (and PC 2's until "PC 2 clear"); the Mac and the laptop update first. PC 2: the aggregation-cost re-run (to about 23:02), then floor-build-4 and floor-sweep-2.
22:44. The coordinator stopped the AMD telemetry agent at about 23:00Z (its clock; the 9070 XT will not be back on the bus tonight); its AMD sweep rows are owed to the morning (relay/playbooks/amd-card-test.ps1, kit amd-kit.zip); it holds no slot. PC 1 is clear for the 0.3.11 update-now once Ember's collect job closes (the shipper reads the intake for no running job on ae432dc7); PC 2's update still waits on the aggregation-cost re-run and the prover-floor build 4 and sweep 2.
## 22:45 the shipper has the tree; the PC queues for the rollout
The shipper took over release-0.3.11 at cc72f4a (b968ee0 + the bump 21173c4 + one CI commit): three of the tree's own checks had failed and are fixed there (the identity check's "MacBook" pattern matched prose in card-lifetime and proving-methods, reworded; the C32 kit-path check flagged the three tools/proving-v1/pc2-*.ps1 scripts for a bare jobs\ literal, now Test-Path'd; the socket check flagged tools/amd-prove/pc1-cpu-prove.ps1 and its -sp sibling, now allow-listed with the reason); every check green (identity 0 of 220, bash-body 15 of 28, kit-path 14 of 14, socket, markers, workflow shell, relay 17, UI 23). The reviewer's C34 follow-up is with the shipper: packaging/mac/packaged-config.sh line 31 must carry the nine-field object before the DMG and the Windows inputs (a fresh install would otherwise start on the four-field digest). Its order: the Mac node build and the two digest readings, the PC 1 build job (node and app, Linux and Windows), the PC 2 suites from the release worktree, the seed's Linux cross-build, the workers, the app, the DMG, the inputs, the push and CI; then the manifest and the machine order.
PC 1: yours to the shipper now (no running job; Ember's run was "aborted (the app is quitting)" at 22:31:06Z, cause unexplained: no 0.3.11 action existed then; the shipper's PC 1 job output may say). PC 2: agg-cost-pc2-3 runs 22:41:15Z to about 22:59Z (PC 2's app log; the intake lags), then the shipper's suite job, then the prover-floor build 4 and sweep 2, then "PC 2 clear" for the update-now. The AMD telemetry agent is stopped (its rows owed to the morning); Ember's re-run after PC 1 shows 0.3.11 (its engine 054e041, the fetches republished).
22:48. C35: tonight's two unattributed app quits share a shape (PC 2 at 20:01:09Z, 20 s after the efficiency sweep's elevated helper was cancelled; PC 1 at 22:31:06Z, 45 s after Ember's job started a second engine beside the installed app), both killing a measurement in flight, both "quit:" lines naming no source. Ember's re-run is HELD until the cause is named: the Ember owner reads the engine's quit path for every caller (the single-instance lock, the API port bind, the helper protocol, the jobs runner's quit command, the updater) and the line before each quit in both PC logs, gives the quit line its source, and makes the second engine unable to make the installed one quit; one line goes to the 0.3.11 tree through the shipper, else the next cut. The AMD agent's last probe (22:45:01Z): the 9070 XT absent by the PnP list 14 minutes after Ember's ADLX read saw it on bus 98; the link flaps on a scale of minutes; its kit amd-kit-2 is on PC 1 for the morning's window.
## 22:48 the ship's first step report: N4 = N5 = 154,800; the workers built (G5)
Tree release-0.3.11 23bc2b2 (cc72f4a plus the nine-field packaged line). N4 = N5 = 154,800 (DAA 136,967 at 22:45Z at 0.965 blocks/s: the publish near 140,200, tip + 14,400 near 154,600, the first multiple of 3,600 at or above it; the 10,800 floor holds until DAA 144,000, about 00:45Z, else re-pinned). The two Windows workers from this tree: igneum-worker-cuda.exe 2b3b8c92... (1,536,512), igneum-worker-opencl.exe edc4a75d... (478,208), both with the resource block, different from 0.3.10's pair. PC 1's build job build-20261005-224654 (node 89dfcb95, app 23bc2b2, Linux and Windows). The app suite 113 + 27 + 8 and igneum-pow 53 + 4 + 19 + 7 green on the tree. In flight: the fork's Mac node and its two digest readings, the seed's Linux cross-build, the prover build and fixtures; then the inputs, the DMG, the push and CI; PC 2's suite job on "PC 2 suites go" (after the aggregation-cost close at about 22:59).
22:50. Two corrections from the reviewer's log reading, for the morning. (1) PC 2's app quit at 20:01:09Z was an update: its log shows "job update-now-0310 (update-now) starts: 0.3.10 is published: re-read the manifest and install now" at 20:01:04Z, the 49 MB download and the installer start; so a 0.3.10 manifest and an update-now job reached PC 2 at 20:01Z, ninety minutes before the fleet publish at 21:39:59Z; the shipper is asked which publish and jobs file that was and whether it was the same build (an unexplained early publish is a release-process question for the morning). (2) C35's order: "node stopped (exit Some(1))" is written by the engine's stop_node inside the quit path, so on both PCs the node's exit is a consequence of the quit, not its cause; the installed app's Quit comes only from the host's close or /api/quit; for PC 1 at 22:31:06Z the open question is whether Ember's playbook sends /api/quit to the installed app before starting its second engine (the reviewer reads the playbook; Ember's re-run stays held).
22:50. WITHDRAWN, the 22:50 entry's point (1): the update-now-0310 lines on PC 2 are at 21:49:24Z (the fleet update), not 20:01Z; the shipper's "nothing published before 21:33Z" stands, and PC 2's 20:01:09Z quit stays unexplained (a cancelled administrator prompt at 20:00:49Z, a PermissionDenied at 20:00:56Z, then a quit with no source; a next-cut item). C35 narrows to Ember's playbook: relay/playbooks/ember-tune-pc1.ps1 lines 125 to 127 POST api/quit to the URL file in $u when its budget is spent; if $u resolved to the installed app's URL file and the budget check fired at once, the playbook quit the installed app 46 s in. The Ember owner confirms before any re-run; the fix would be that the playbook never addresses the installed app's URL file (its own scratch app dir only) and never calls /api/quit on a URL it did not create.
## 22:52 the digests and the DMG; PC 1's app down since 22:31 (the fleet's hash rate and the ship's PC 1 step)
The shipper's second report (tree 23bc2b2): the two digests on the 0.3.11 Mac node: c562d70e... with no override file (= the node agent's pinned test), 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 with the nine-field object at N4 = N5 = 154,800, the value every node must print after the publish (the node logs the class switch "active from epoch 43" and the proving v1 line); the DMG b7e81d4f... (41,592,041 bytes, the nine fields read back from the image); the seed's Linux node 63cf490d.... Waiting on PC 1's exes (build-20261005-224654) for the inputs, the pin, the push and CI.
PC 1 (Ember's finding, confirmed on the intake at 22:51): the installed app has not come back since its quit at 22:31:06Z (the last upload 22:31:08Z, no new run id, no job since), so PC 1 is not mining (the devnet short its 141 MH/s for 20 minutes), the shipper's build job cannot start there, and no update-now can reach it. The relay agent on PC 1 is alive (read-only probes ran through it at 22:45); the shipper is asked to relaunch the installed app through a relay task in the interactive session and to verify a new run id; if the relay cannot reach the user's session, PC 1 waits for the project lead in the morning and the 0.3.11 rollout goes without it (its update lands at its relaunch). the project lead is not woken. The quit's cause (C35): the senders are the tray Quit, stdin EOF in wrapper mode and POST /api/quit with the token; Ember's playbook POSTs api/quit to the URL file in $u when its budget is spent (lines 125 to 127); the Ember owner is checking what $u resolved to at 22:31; Ember's re-run stays held, and the playbook rule becomes: never read the installed app's URL file, never POST quit to a URL it did not create.
22:53. PC 1: the relaunch of the installed app went out through the relay at 22:52:31Z (item #241: igneum-app.exe started through explorer.exe so the app gets the user's own token, a one-shot scheduled task at limited run level as the fallback); the shipper reports the new run id and the first STATUS line; its build job starts when the app fetches the jobs file. PC 2 queue after the aggregation-cost job (closing about 22:59): the shipper's 0.3.11 suites, the prover-floor build 4 and sweep 2, the aggregation-cost re-run (20 min, with a plain-text parser: under 0.3.10 the app's /api/state comes back empty to PowerShell 5.1's JSON reader while the body is there, a class hit by three playbooks tonight; the app owner fixes the response shape or every playbook parses the text), then "PC 2 clear" for the update-now, then the ledger-pc2 agent's M16 / E17 / P17 job (the inline-cache kernel at 64 and 256 MiB on the 5090 against the honest kernel, the nvidia-smi line per setting, the igneum-exec suite in WSL2; about 12 minutes, after its kit lands on the dl folder in 1 to 2 hours).
22:54. C35 resolution (Ember owner): $u resolved to %LOCALAPPDATA%\igneum-tune\app\app.url, the second engine's own file (lines 34 to 36 and 125 of the playbook; line 74 deletes it before the engine starts; platform::data_root() honours IGNEUM_APP_DATA, set to the scratch root at line 93), and the budget branch never ran (its "RESULT TUNE error=budget_exceeded" line is absent); so the playbook did not quit the installed app. What the reading did find: the second engine counted --sweep as Power control and raised a UAC prompt at about 22:30:25Z (apply_power_limits at start), 40 s before the installed app's quit; closed on ember-tune b671c8b (Power control alone decides, no cap at start under --sweep, every Cmd::Quit names its source, the budget floored at 5 minutes, the playbook refuses a quit to any URL under the installed igneum\app). The remaining question is whether an unanswered elevation prompt can take the installed app's window host down (stdin EOF): the same shape as PC 2's quit at 20:01:09Z, 20 s after a cancelled administrator prompt (C16). "go PC 1 collect" given: the Application event log at 22:31Z and the installed app's log tail, once PC 1's app is back; the re-run stays held until the source is named.
22:55. C35 narrows to one common factor: both unexplained quits came 20 to 41 s after an administrator prompt beside the running installed app (PC 2 at 20:00:49Z then 20:01:09Z; PC 1 at about 22:30:25Z then 22:31:06Z); Ember's playbook is ruled out. Rule for the night (rollout plan 4a): no PC job raises an elevation prompt on either PC; the two-minute class test (one prompt raised and cancelled beside the mining app, the quit line read) waits for the morning with the project lead present, or tonight only after the relay relaunch path is proven and the rollout is done; Ember's quit-source stamping (b671c8b) goes to the next cut. The devnet has been short PC 1's 141 MH/s since 22:31Z (96 MH/s at 22:51); the relay relaunch #241 is the recovery, else PC 1 is on the morning's hands list beside the eGPU reseat and the 16:00Z check.
22:55. PC 1's engine did not exit: relay task #241 found igneum-app.exe ALIVE (pid 26696, started 21:49:40Z, session 1) answering nothing on /api/state in 60 s and uploading nothing since 22:31:08Z: it logged its quit at 22:31:06Z, stopped the miners and the node, and hung instead of exiting (a quit that never ends: a C35 fact and a next-cut defect: the quit path must end the process or the watchdog must end it after a bound). Task #243 (22:55:10Z) ends that engine by pid, as the app's own updater does, then starts the per-user install through explorer.exe (the user's token), the limited-run-level scheduled task as the fallback; the new run id, the first STATUS line and the build job's first STAGE line follow in the intake.
22:56. fud-close (the ledger closer's 45 public-text fixes, two CI checks, the relay fixes; 0 conflicts with ca2-coord) is added to the ship order after ca2-coord if its ready tip reaches the shipper before the inputs are pushed (the workers and the DMG rebuilt from the merged tip, G5); else it heads the next cut. The fork-side ledger-fixes (from release-0.3.6) is not in 0.3.11.
22:57. Correction: fud-close is NOT in 0.3.11 (the tree closed at 23bc2b2 before it reached the ship order; no late branch, the 0.3.10 rule); it heads the next cut with the fork-side ledger-fixes rebased onto the 0.3.11 fork. The push and CI go the moment PC 1's exes land.
22:57. Next-cut coupling recorded (the ledger closer): fud-close 647b08c's worker half of M28 (the kernel_sha256 check in packfile.h and host.c) must ship with the fork-side ledger-fixes miner commit 3d4ec451 that stamps the hashes, or every worker refuses every pack; ledger-fixes is not yet rebased onto 89dfcb95 (two conflicting files); round 2's ledger branches stay behind tonight.
22:59. Next-cut note from the reviewer's merge-tree against 23bc2b2 (0 conflicts for explorer d7e797c, ember-tune b671c8b, rig-install 086008a, pool-v0 425b875, ota-k2 89a76b1, hive-words 2d056e8, fud-close 647b08c and the others already in): explorer d7e797c makes tools/ci/public-api-check.mjs fail when the live /api/stats lacks `proving`, and that check runs on master pushes against the live site, which Vercel redeploys only after the push, so the first master CI after a ship carrying it goes red through no fault; the next cut holds d7e797c or gives the check a retry loop. Waiting now on two watchers: the aggregation-cost close on PC 2 (then the shipper's suites) and PC 1's app relaunch (then the exes, the push and CI).
## 22:59 PC 1: the hang explained; orphaned miners from the second engine hold both GPUs
The shipper's relay task #244 (22:55:24Z) counted 1 igneumd, 2 igneum-miner, 2 igneum-worker-cuda and 1 igneum-worker-opencl running although the installed app had stopped its miners at 22:30:20Z and its node at 22:31:06Z: Ember's second engine's children, orphaned when its job was aborted, mining on both GPUs; they would fight the relaunched app's miners and void every number. The relay lane kills the tree (taskkill /F /T on every miner, worker and non-app igneumd) and relaunches the app. The hang (Ember's reading): the installed engine's quit got stuck in the jobs runner's abort, whose reader waits for EOF on the script's stdout pipe; the pipe's write end was inherited by the second engine and its miners (PowerShell's Process.Start with redirection inherits every inheritable handle), so EOF never came and the engine sat "responding" until #243 ended it. Class rule for every playbook that starts a second engine (ember-tune-pc1.ps1, relay/playbooks/sweep-5090.ps1 and any job script of the shipper's): no inherited pipe into the second engine, its whole process tree killed at the end and on abort, the installed app's miners restarted only after; a CI check that fails a playbook starting an engine without those lines (Ember's branch). The quit's sender is still open (the tray excluded by the missing 45-s host timer kill: stdin EOF or POST /api/quit; the event-log collect decides, when the app is back).
23:00. The second-engine rule is closed on ember-tune 8ab9068 (docs/plans/ember-tune.md section 5; the playbook pair fixed; tools/ci/second-engine-check.sh in ci.yml, shown to fire on a known-bad playbook and pass the fixed pair); next cut. The shipper's relay kill-and-relaunch task on PC 1 goes out at about 23:03:30Z (or the kill alone at once if the new engine reports first); the build job follows on a clean PC 1.
23:03. C38 (the reviewer): the documents of ca2-analysis ee42d7c (sram-mirror.md, int8-matrix-family.md, the dot4 probe sources), ca2-soundness a465881 (scratch-soundness.md; its scratch tests reached ca2-v3 through the mixer branch), ca2-epoch e95e8b5 (epoch-length.md) and prover-floor (prover-floor.md) exist on their branches only: my ship list omitted them, and evidence rows 17 and 19, the litepaper's chip bullet, chip-model-v3.md and the rollout plan's layer 9 row cite them. The shipper decides before the push: a docs-plus-standalone-sources merge of the four (0 conflicts for docs/, nothing the gates ran on, the shipped code paths' diff verified empty), or, under the 0.3.10 no-late-branch rule, the citations changed to "on branch <name>" on ca2-coord as one docs commit and the four heading the next cut beside fud-close.
23:05. C38 closed on ca2-coord: ca2-analysis ee42d7c, ca2-soundness a465881, ca2-epoch e95e8b5 and prover-floor cfe3d80 merged (8c00f84, 271cd63, f940101, ead67e0; the bench log both sides each time), the soundness branch's two code files set to the release tree's versions (0b505f9; an empty diff against 23bc2b2), so the five cited documents are on the branch the ship takes and its merge is one docs commit; the shipper decides whether at 0.3.11's master merge or the morning's cut.

View file

@ -16,6 +16,21 @@ Trigger: the 9070 XT measurement of 5 October (bench-log "the 9070 XT on the eGP
| 6 | Cache growth on the genesis schedule | the SRAM mirror stays unaffordable | none | already in the design; confirm the schedule against SRAM density |
| 7 | Integer matrix ops in the program (INT8 x INT8 into INT32, exact) | matrix hardware at GPU scale | none on NVIDIA and AMD; Apple to check | reserved family, not at launch |
| 8 | Working-set size drawn per program | one memory design cannot fit every hour | none | folds into 4 and 5 |
| 9 | Epoch length as a signalled era parameter: base 3,600 DAA s, ladder 600 to 7,200, set by 90% signal at a day boundary, lead and `T_epoch` fixed | a per-program bitstream (an FPGA with a hard datapath): at 600 s nothing it compiles ever runs (42 to 160 min per compile, PRflow FPT 2019; hours on large parts, Aldec) | compile-ahead 1 s per epoch on the 5090, 0.5 s on the M5 Max (38 s with the race on); one CPU core `600 / epoch_len` busy on the VDF | reserve-only tonight: `docs/plans/epoch-length.md` |
## Decided 5 October 2026 (night), under the project lead's delegation for the devnet (`docs/plans/counter-asic-2-rollout.md` section 6)
| # | Decision | The number that decided it |
|---|---|---|
| 1 | keep v2's 128 x 4 B | w16 passes the rule but closes nothing (5090 139.8 against 136.1 MH/s, 9070 XT 17.90 against 18.15); w64 makes the 5090 bandwidth-bound (share 0.58, 37% of stream) |
| 2 | out | spread across six programs 5.5 to 22.3% per card, over the 5% rule |
| 3 | out (scratch share 0); the construct is sound and its tests stay | the on-die-cache recompute chip stays at 2.4x at every share under the 6 GB cap |
| 4 and 8 | IN: the era draw of stride, interleave and the working-set window (width pinned at 4 B) | six-era spread 1.3% on the 5090, 3.2% on the 9070 XT, 0.8% on the M5 Max; bit-exact on all three vendors |
| 5 | OUT of v3 (a measured option for 3.0) | the added form costs the 5090 13 to 16% and the 9070 XT 16 to 20% (g 0.87 / 0.85 / 0.84 and 0.84 / 0.81 / 0.80 at 32 / 64 / 96 MiB against the 0.97 rule): no card keeps even 32 MiB resident while the dataset streams; the replaced form helps the on-die-cache chip (x1.33 at k = 4) |
| 6 | option C: the cache doubles when the dataset doubles | the mirror is 128 mm^2 and $46 at N5 by shipped density; the cache's job is to stay above GPU L2 |
| 7 | reserve R1 = mm8, unsigned, W_new 4, unlock era 4 or 90% signal | native on all three vendors as a tile; dot4 emulation 1.6x on Apple |
| M16 mixer x8 (x4 measured beside it) | into v3 | the only measured lever that moves the named chip: x8 0.31x bare, 0.92x with a 3x fixed-function factor (x4: 0.61x, 1.84x); verifier 2.1x v2 per warp (2.79 ms on a loaded core, about 1.3 ms quiet); the daily build unmoved on every card (latency-bound) |
| 9 | reserve-only, no change to the devnet's hour | the floor 600 s from the slowest compile-ahead (38 s) |
Not added: divergent data-dependent branches (cost GPUs more than chips), anything floating point (bit-exactness across vendors).

View file

@ -0,0 +1,30 @@
# Counter ASIC 3.0
The third set of chip-resistance layers, from the ASIC-history agent's audit of 5 October 2026 (`docs/analysis/asic-resistance-history.md`, branch asic-history: 31 chip histories with the gain per joule, the months held and the response; the chip economics; the audit of class v2 and of every Counter ASIC 2.0 layer). The rule is the 2.0 rule: every item is measured the same way (the three cards we own, the chip model per variant, bit-exactness, the verifier cost), what passes is folded into the class as v4 behind its own activation (`program_class_v4_activation_daa`, by DAA height like v3), under the same six gates and the same rollout shape (`docs/plans/counter-asic-2-rollout.md`). Nothing here is active; nothing touches the devnet until it has its measurement and the project lead's word. Written at 22:4x UTC on 5 October 2026, after the 2.0 class was decided and before its publish.
## The ranked additions
| # | Addition | What it takes from a chip | Where it sits | Step |
|---|---|---|---|---|
| 1 | The partial-store chip and the time-memory curve: price a chip that holds a fraction f of the dataset (f = 0.25, 0.5, 1) on HBM3 or GDDR7 with 4-byte access granularity and recomputes the rest, scored in energy per hash | the only chip class that beat a memory-bound GPU hash (Ethash: 2.1x Linzhi 2020, 2.9x E9 2022, 4.8x per joule Jasminer X4 2021) did it with custom memory controllers and on-package memory, not an on-die dataset; chip-model-v3.md prices only f = 0 | `docs/analysis/chip-model-v3.md`, O-1.6, MEMHARD.md section 3 item 2 (the curve never drawn) | analysis, before the public testnet's vectors freeze; the first item |
| 2 | A random item-derivation program per day in place of the fixed-shape mixer (RandomX's SuperscalarHash idea) | the fixed mixer shape IS the 3x fixed-function allowance that turns x8's 0.31x into 0.92x; removing it is worth more than x16 (0.46x with the factor) | a reserve family now; genesis if the per-day compiled derivation verifies under the 10 ms gate (unmeasured); risks: cryptanalysis of random ARX, weak draws, bit-exact compilation on three vendors; the daily build about doubles (23 to 77 ms, approximate) | design and the verifier measurement |
| 3 | External cryptanalysis of the mixer M_r, the chained cache and the acceptance rule, with the x8 shape as the target | MTP fell from 2 GB to under 1 MB before launch (Dinur and Nadler 2017), Catena's proofs were flawed, Argon2i's parameters were attackable; RandomX bought four audits for about $141,000 before launch; x8 multiplies the mixer's weight in the chip model, so a structural shortcut is worth 8x more | ledger M7, raised to a genesis gate | commission before genesis |
| 4 | The clock and the detector: (a) a share-pattern detector on the observer (per-program hash-rate spread, nonce-group patterns, per-card-model rate bands; alert when a population behaves like one fixed design: how MoneroCrusher found Monero's secret chips at 85% of the hashrate, February 2019); (b) the bounty's trigger as daily issuance in dollars, not a date (compute-bound hashes got chips at $20K to $30K a day: Radiant, Kadena, Handshake; Vorick's 2018 rule about $55K a day) | not a layer: the response time | the observer (`tools/observer`), D11 (the bounty is unfunded) | before the public testnet |
| 5 | Rank layer 9 (the epoch length) above layer 7 and measure the FPGA lane: a soft-overlay FPGA with HBM (reads in flight per watt against the 5090's 17.5 G/s) added to the compile-ahead measurement | FPGAs were the first adversary of Lyra2REv2 (2018) and X16R (1.3x, September 2019) and came back within weeks of X16Rv2; Xelis forked for FPGA resistance (July 2024); a per-hour compiled program is a bitstream target | `docs/plans/epoch-length.md` | measurement before the public testnet |
| 6 | Order the reserve by chip-unfriendliness: the 32-bit datapath families first (byte permute, bit-field extract, variable shifts, popcount, select, the second shuffle), mm8 last | int8 matrix blocks are licensable IP at every node; Apple pays 1.6x to 4.7x per emulated dot4; Least Authority's ProgPoW suggestion 5 was "watch ML hardware" | spec 1.13.2 | a decision for the project lead with the 3.0 measurements |
| 7 | A vendor-share metric (hashrate by vendor) published with the benchmark | the 7.5x AMD gap is a softer form of the capture the history records (Kaspa's GPU share went to nothing within months of KS0) | the numbers page, the observer | with the public benchmark |
Placed nowhere, with the reasons in the history document's section 4.3: per-hash programs (the 25x GPU penalty RandomX pays), Verthash's table-from-chain, Grin's dual PoW, Autolykos non-outsourceability, a per-hash VRF (NexaPow's got a 3x to 4x chip), ternary or variable-precision ops, cache-timing reads (measured out as layers 3 and 5), branches and floating point (excluded).
## Decisions for the project lead raised by the history
Add the partial-store rows before the vectors freeze; name the random derivation as a reserve family and fund its verifier measurement; commission the mixer cryptanalysis; escrow the bounty on an issuance trigger and build the detector; rank the epoch-length reserve above mm8.
## The plan, in order
1. Item 1 (analysis): the partial-store chip rows and the time-memory curve in `docs/analysis/chip-model-v3.md`, with the sector size per card (the 5090 moves a 32-byte sector per 4-byte read, the 9070 XT 64) and the HBM3 and GDDR7 random-read rates cited; the first item because it can move the public claim.
2. Item 2 (design and measurement): the per-day derivation drawn from a reviewed fixed set, the verifier cost on one core, the daily build on the three cards; reserve entry text for 1.13.2.
3. Items 4 and 5 (tooling and measurement): the detector on the observer, the issuance trigger, the FPGA overlay estimate.
4. Item 3 (procurement): the cryptanalysis brief and the target shape.
5. Items 6 and 7 (decisions and the benchmark page).
6. What passes, as class v4 behind `program_class_v4_activation_daa`, the six gates, the rollout shape of 2.0.

181
docs/plans/ember-tune.md Normal file
View file

@ -0,0 +1,181 @@
# Ember Tune: every card tuned for MH per watt, out of the box
5 October 2026, night. the project lead: "make sure we have ember tuning every single card for efficiency out of the box, the
more data = the better the tune, make an awesome system." Branch `ember-tune`, worktree `../igneum-wt-ember-tune`,
on top of the AMD telemetry commit (7adcd4c, branch `opencl-rdna4-telemetry`) and the Power control commit (3562f26,
branch `job-console`), both cherry-picked. Lever 3 of docs/plans/miner-eff.md grows two knobs and a fleet memory;
lever 2 (docs/design/miner-tuning.md) carries the priors in the same signed `tuning` section.
## 1. What a user sees
| Moment | The card row says | What happened |
|---|---|---|
| First 2 minutes of mining | `tuning: waits for 120 s of steady mining` | The worker warms up; nothing is touched. |
| Tuning | `tuning: holding 2472 MHz · 100% · 41 s (step 7 of 9)` with a Stop button | One card at a time, on the live kernel, never restarting the worker. |
| Tuned | **Tuned: 122.3 MH/s at 290 W (0.422 MH/W)**, then `2470 MHz at 100%, full tune, 1 h ago` | The point is pinned on the card; the result went to the fleet. |
| Known model | the same line, `from the fleet prior, confirmed, 2 min ago` | The card started at its model's prior and confirmed it in two steps instead of nine. |
| Apple silicon | **Tuned: 26.7 MH/s at 38 W (0.703 MH/W)** `(measured as it runs)`, and `measure only on Apple silicon: the system sets the clocks and the power; no control exposed` | Nothing can be set; the number is still reported so the row and the fleet know what the card does. |
| NVIDIA, Power control off | the measured line and `measure only until Power control is on in Settings (Windows asks for administrator rights once)` | The app never raises the prompt by itself (5 October 2026). One switch, one prompt, and the full tune runs. |
| Slider moved | `your setting stays pinned` | A manual point is never overridden; the tune still measures and reports. |
| Stopped | `tuning stopped: a remote job took the GPU` and the card back where it was | Any fault reverts the step and the run. |
| Fleet pause | Settings: `tuning paused fleet-wide by the signed manifest` | The kill switch. |
Settings: one switch, "Ember Tune: tune every card for MH per watt out of the box (once after install, then weekly,
and after a driver or program change)". AMD needs no rights. NVIDIA needs the Power control switch (one administrator
prompt) for both knobs; off, it measures only.
## 2. The knobs, per vendor
| Vendor | Power limit | Core clock cap | Memory clock | How | Rights |
|---|---|---|---|---|---|
| NVIDIA | `nvidia-smi -pl <W>`, percent of the default, inside `power.min_limit` and `power.max_limit` | `nvidia-smi -lgc 0,<MHz>`, percent of `clocks.max.gr`; `-rgc` = unlocked | never touched (`-lmc` is not used); read back as `clocks.mem` | directly when the engine is elevated, else the one-prompt helper (`<seq> pl <W>`, `<seq> lgc <MHz>`, `<seq> rgc` in `sweep/cmd.txt`) | administrator, so only with Power control on |
| AMD | `igneum-gpu-telemetry --card N --set-plimit <offset>` (0 = default, -20 = 80%), inside the `tune` line's `plimit_range` (PC 1's 9070 XT: -30 to 10, so 70% is the floor) | `--set-gmax` only when the `tune` line's `gmax_range` is absolute MHz (floor 0 or above); on RDNA 4 the range is an offset from stock (-500 to 1000 on PC 1) and the clock knob stays closed until the stock clock is known; `--reset` for the default point | not settable through ADLX on RDNA 4; read back as `mclk_mhz`, and a step whose mean memory clock falls under 95% of the baseline's is marked and cannot win | the helper, one process per request, exit 0 and a `tune ... ok` line | none on Windows (ADLX manual tuning); root on Linux, so measure only there |
| Apple | none | none | none | measure only | none |
Vendor limits are never exceeded and the floor is never undercut: the plan clamps every point (`Limits::clamp_clock`,
`Limits::watts_for`), and a clock floor the vendor does not report is 60% of the maximum.
## 3. The plan and the choice
Full plan (a new model, or a prior that lost its confirm check): the power ladder 100, 90, 80, 70, 60, 50% at the
unlocked clock (duplicate watts dropped where the card's floor clamps them), then the clock ladder 90, 80, 70, 60%
of the maximum at the power point the power ladder chose. 60 s hold after 15 s settle per step; 9 steps on an
RTX 5090 (five power, four clock), about 12 minutes.
Confirm plan (the model's prior has 5 or more reports): the prior's point, then one neighbour (the next clock step up
when the prior caps the clock, else one power step down). If the neighbour beats the prior by over 1% MH/W, the full
plan is queued; else the prior stands. Two steps, about 3 minutes.
Baseline plan (measure only): one step at the card's current point. The "before" number for the row and the fleet.
The choice (`ember::choose`): among the usable steps whose rate is within the tolerance (1%, settable from the
manifest) of the fastest step, the best MH per watt; within 1% on efficiency the higher rate; within 1% on both the
lower draw. A card never gives up more than the tolerance in blocks for the saving. A step is unusable when it is
marked: `faulted` (a rejected or mismatched hash during the hold: the step is reverted and marked), `hot` (the GPU
reached 85 C; the run aborts at 90), `memory_clock_dropped`, `unapplied` (the readback disagreed with the request),
`no_readings` (under three draw samples or no STATUS line).
## 4. The data flow
```
card mines 120 s ──> probe (limits, driver, how to set) ──> plan ──> steps ──> choice ──> point pinned
│
app log: TUNE start / TUNE card=.. step=.. / TUNE chosen / TUNE {json} (and stdout under --sweep)
│
log upload (every minute, the existing intake, site/api/log.mjs) ──> Neon miner_logs
│
relay/lib/ember.mjs aggregate: per (card model | driver major | program class)
median clock cap (10 MHz), median power %, median MH/W, MH/s, W, spread (MAD %), samples, machines
│ │
console: /r/<token>/c/tuning, `node tools/console.mjs tuning` site: tools/tuning.mjs --priors --site
│ -> site/miner-priors.json -> /miners#priors
tools/tuning.mjs --priors --write tuning.json (priors + ember settings beside the kernel-variant cards)
│
packaging/ota/publish-manifest.sh --tuning tuning.json --deploy (signed; carried over when not given)
│
every app: <app data>/tuning.json ──> ember::settings_of (kill switch, min samples, tolerance, period)
──> ember::prior_of(key) ──> a new card's confirm plan
```
The record (`ember::record_json`): `ts`, `machine` (the first 8 hex of SHA-256 over the install id; the id itself
is random per install and never sent), `app`, `os`, `card`, `vendor`, `driver`, `driver_major`, `class`, `key`,
`plan`, `steps` (the full table: clock, power %, limit, watts, MH/s, MH/W, core and memory clock, hottest reading,
faults, mark), `chosen`, `before` (the full plan's 100% step), `eff`, `mhs`, `watts`. The key: `<card model with
underscores>|<driver major>|<program class>`, the class from the worker's race line (`l128w16` today; `v2` before a
race has run).
## 5. Scheduling and safety
| Rule | Where |
|---|---|
| One card at a time; the card must have mined 120 s and have a STATUS line | `tick_sweep` |
| Never under a remote job hold, a pause, inside 600 s of the hour boundary, or while the app quits | `tick_sweep`, `sweep_drive` |
| Due once after install, every 7 days (manifest `ember.period_s`), and when the driver major or the program class changed since the last tune | `tick_sweep` (`CardPref.sweep_driver`, `sweep_class`) |
| A pinned card (the slider) is measured, never changed | `sweep_finish` |
| Kill switch: `tuning.ember.enabled = false` in the signed manifest stops every tune fleet-wide; the Settings line says so | `ember::settings_of`, `tick_sweep` |
| Faults: a rejected or mismatched hash marks the step; the card leaving `mining`, a worker error, a job, a pause or 90 C aborts the run and restores the point from before | `Run::sample_fault`, `sweep_drive`, `sweep_abort` |
| Memory clock held: never set; a step that drags it under 95% of the baseline's cannot win | `Row::from_samples` |
| Vendor limits: every point clamped to the reported range; the clock floor 60% when none is reported | `Limits` |
| A signed prior is only ever a starting point inside the card's OWN reported limits (`power.min_limit` to `power.max_limit`, the clock floor to `clocks.max.gr` or the ADLX `gmax_range`), never a memory clock, never a value the card did not report; the confirm step measures it and the full plan replaces it when a neighbour beats it, so a bad prior costs the fleet one confirm step per card, not a setting. The signing key (K1, docs/security/keys.md) therefore cannot push a card past its vendor ceiling or under its floor | `Plan::confirm` clamps through `Limits::clamp_clock` and `power_pct.clamp(50, 100)`; proven by `ember::tests::the_confirm_plan_checks_the_prior_and_its_neighbour` (a prior of 9,000 MHz at 30% becomes 3,090 MHz at 50%) and `limits_never_exceed_the_vendor_or_undercut_the_floor` |
| No prompt the user did not ask for: the NVIDIA helper starts only with Power control on; the `--sweep` job never counts as permission | `sweep_probe_known`, `sweep_helper_start` |
| The elevated helper restores the limit and resets the clocks by itself after 20 idle minutes | `sweep::helper_script_*` |
| A playbook that starts a second engine beside the installed app (the PC measurement jobs) gives it NO pipe (its output goes to a file the script tails: a pipe's write end is inherited by the engine's miners, and the installed app's jobs runner then waits forever for EOF after an abort; C35, PC 1 22:31 UTC, a 24-minute hang and orphaned miners), ends the engine's whole process tree at the end and on the budget (`taskkill /T /F`), and lets the installed app's miners come back only after that | `relay/playbooks/ember-tune-pc1.ps1`, `sweep-5090.ps1`; CI `tools/ci/second-engine-check.sh` fails any playbook without both |
| Every `quit:` line in the app log names its source (the window host's stdin, the host gone, `POST /api/quit`, the `--sweep` run's end) | `Cmd::Quit(&'static str)` (b671c8b) |
| A second engine never runs the updater: `IGNEUM_APP_NO_OTA=1` (implied by `--sweep`) skips the OTA tick and refuses Check now, whatever the manifest's `min_supported_version` says (the installer it would launch quits the installed app: PC 1, 22:31 UTC) | `Engine.no_ota`; the playbooks set the variable; `tools/ci/second-engine-check.sh` demands it |
## 6. Tests
| Test | What it fixes |
|---|---|
| `ember::tests::the_full_plan_is_the_power_ladder_then_the_clock_ladder_at_the_chosen_power` | 5 + 4 steps on the 5090's limits, the clamps, the dynamic second half, the 1% and 5% choices |
| `limits_never_exceed_the_vendor_or_undercut_the_floor` | clamps |
| `the_choice_keeps_the_best_mh_per_watt_within_the_rate_tolerance` | the rule, the ties, marked rows never win |
| `the_guards_mark_a_step_so_it_cannot_win` | faulted, hot, memory clock, unapplied, no readings, the line |
| `a_fault_during_a_step_reverts_it_and_the_run_goes_on` | the state machine with a fake clock: the faulted 70% step is marked and never chosen |
| `the_confirm_plan_checks_the_prior_and_its_neighbour` | the two steps, Keep against FullDue, a prior outside the range clamped |
| `a_baseline_plan_measures_the_card_as_it_runs` | no control, still a number and the Tuned line |
| `the_record_and_the_prior_round_trip_through_the_manifest_shape` | record fields (no address, no host), `priors` and `ember` beside `cards`, the sample floor, the kill switch |
| `control_reasons_per_vendor` | who measures only and why |
| `sweep::tests::helper_scripts_carry_the_protocol` | the helper's `pl`, `lgc`, `rgc` |
| `relay/test/ember.test.mjs` | five samples converge (2,470 MHz at 100%), an outlier (0.908 MH/W at 1,854 MHz) moves nothing, baseline records make no prior, de-duplication, the manifest merge keeps lever 2's cards, the canonical round trip, AMD keys |
| `app/igneum-app/ui/tune-line.test.mjs` | the row line per state |
Run: `cargo test -p igneum-app ember sweep` (on a PC through the build job, or on the Mac under the build lock),
`node --test relay/test/ember.test.mjs app/igneum-app/ui/tune-line.test.mjs`.
## 7. The tier consequences
| Tier | What Ember Tune does for it | What it costs |
|---|---|---|
| A laptop GPU (NVIDIA, 60 to 115 W) | the power ladder usually finds the vendor floor binding; the clock ladder is where a memory-bound program saves watts; the thermal mark keeps a hot chassis from winning a step it cannot hold | about 12 minutes once, then 3 minutes a week; under 1% of the hour during the tune (the worker never stops) |
| One 8 GB card | the same two knobs; the 8 GB card is identities-limited (2 by default), the tune does not change that | the same |
| One 12 or 16 GB card | the same | the same |
| One 24 or 32 GB card (the 5090) | the draw sits far under the cap (290 W under 460 W on PC 1), so the power ladder is flat and the clock ladder is the lever; expected saving from the 4 October stability line: tens of watts at under 1% rate, to be measured | the same |
| A rig (several cards) | one card at a time, so a six-card rig takes about 70 minutes to tune once; every card of one model after the first starts at the prior (3 minutes); the tune never touches a card a remote job holds | linear in cards once, then the confirm plan |
| A pool user | the same per card; a pool submits the same hashes, so the 1% rate tolerance is the same 1% of shares | the same |
| AMD on Linux | measure only (sysfs needs root); the row says so | 60 s a week |
| Apple silicon | measure only; the row says so | 60 s a week |
Privacy line: what is uploaded is the record in section 4 and nothing else: a hash of the random install id, the
card model, the driver version, the OS, the program class, the step table and the chosen point. No address, no
hostname, no raw machine id, no user name. The public priors table carries only the aggregate per model.
## 8. Measurements
### PC 1, 5 October 2026 (night)
Tonight's constraints, read from PC 1's own uploads: the installed app runs as `DESKTOP-KMCV30N\Admin` with
`elevated=False` (the account line at 19:02:33 UTC), the two in-app sweep attempts at 20:09 UTC aborted on the
cancelled administrator prompt (`SWEEP aborted ... the_elevated_helper_did_not_run_(the_administrator_prompt_was_cancelled)`),
so no stored sweep result exists from today, and the RX 9070 XT left the PCI bus at about 20:40 UTC (eGPU link,
not restarted tonight). NVIDIA's `-pl` and `-lgc` need administrator rights, the project lead is asleep, and the app never raises
the prompt by itself, so tonight's run on PC 1 is the baseline plan on the 5090 through the whole pipeline (probe,
measure, TUNE record, upload, aggregation, prior shape in a test manifest). The two-knob tune on the 5090 and the
9070 XT run are owed: the 5090 the moment Power control is switched on (one prompt, then the tune runs by itself
within 2 minutes of steady mining), the 9070 XT when the card is back on the bus.
Run 1 (ember-tune-pc1-1, 22:30 UTC): aborted 46 s in by the installed app quitting, named the next morning: the
second engine's own updater (0.3.9 under min_supported_version = urgent) ran the per-user installer, whose
PrepareToInstall quit the installed app through its api/quit (C35 in the bench log); before any step; nothing set; the "before" snapshots are in the bench log (5090: 450 W of 575, 2,850 MHz core, 3,090 MHz
maximum, 14,001 MHz memory; 9070 XT present on bus 98 with OFFSET ranges `gmax_range -500 1000`, `plimit_range -30
10`). The offset finding changed the AMD mapping (054e041): an offset clock range closes the clock knob and the power
ladder runs on a percent scale bounded by `plimit_range`. The re-run follows the 0.3.11 rollout.
## 8a. Next-cut notes (for the 0.3.12 shipper)
| Commit | What | Where |
|---|---|---|
| b671c8b | every `quit:` names its source; Power control alone decides; no cap at start under `--sweep` | main.rs, server.rs, engine.rs (separable) |
| e600e63 | a second engine never runs the updater (`IGNEUM_APP_NO_OTA`, implied by `--sweep`) | engine.rs (6 lines, separable) |
| 1e9550e (this commit, amended) | the elevated job path's output file is followed while the script runs, so the 5-minute progress reports carry its lines (a 35-minute run that never mined showed only "script running" on 6 October 2026); the tune playbook's watchdog fails a run that mines nothing within 120 s of its first status line, with the engine's last log line in the RESULT | jobrun.rs `follow_file`, relay/playbooks/ember-tune-pc1.ps1 |
## 9. Open
- The NVIDIA clock readback: `nvidia-smi -lgc` is confirmed only through the core clock during the hold (a mean over
the cap by 5% marks the step `unapplied`); the first run with Power control on tells whether the driver honours
the lock on the 5090 under this kernel.
- ADLX on RDNA 4 exposes no memory-clock setter; the memory-clock mark is the guard. The telemetry agent's 9070 XT
sweep tells whether a core cap drags the memory clock on that card.
- The confirm plan's neighbour is one step; a second neighbour (the other knob) would cost 75 s more and catch a
prior that is wrong on both knobs.
- Intel: no knob yet; the row says measure only.

235
docs/plans/epoch-length.md Normal file
View file

@ -0,0 +1,235 @@
# Epoch length as an era parameter (Counter ASIC 2.0, layer 9)
5 October 2026 (night), branch `ca2-epoch`, worker "ca2-epoch". the project lead: "what about faster program changes?". Status: Designed, with one Measured section (the Mac compile-ahead, section 6) and cited figures for the PC cards. Reserve-only tonight: nothing here changes the devnet's 3,600-DAA-s epoch, no code moves, no consensus effect. The parameter joins the era-parameter table of spec 01 section 1.13.1 beside the layer 4 and 8 draws of `docs/plans/era-layout.md` (branch `ca2-era`).
## 1. What the layer is
Today the program changes every 3,600 DAA s (spec 01 section 1.12, "a prototype value, to be fixed at gate 2"). This layer makes the length a genesis-reserved parameter, `epoch_len`, base 3,600, settable by 90% miner signal (spec 05 section 5.7) on a fixed ladder between 600 and 7,200 DAA s, with no fork and no release. The point is a chip that must be rebuilt per program (an FPGA with a hard datapath): the shorter the epoch, the smaller the share of each epoch it can mine. Against a GPU the cost is the compile-ahead cadence, measured in section 6, which is what decides the floor.
| Decision | Choice | Why |
|---|---|---|
| How the length is set | 90% signal only; the era stream consumes one draw for it and ignores the value (section 2.4) | a random length buys nothing against the threat and moves two costs (difficulty settle, VDF duty) at random every era |
| Base | 3,600 DAA s | the devnet's hour, unchanged |
| Ladder | 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 (index 0 to 7) | every step divides the day (86,400) and the era (15,552,000), so day and era boundaries are epoch boundaries at every length |
| Where the schedule anchors | the day, not the era (section 2.2) | a change lands at a day boundary, days after the signal, not 180 days after |
| The seed lead | unchanged: checkpoint at least 1,200 DAA s before the epoch start, `T_epoch` fixed at 600 s on the reference core (section 3) | the known-program window stays 600 s at every length; the grinding margin and the checkpoint depth are untouched |
| The floor | 600 DAA s (section 6.3) | the slowest measured compile-ahead (the Metal race, 38 s) is 6.3% of 600 s and inside the 600-s window |
## 2. The parameter
### 2.1 Definition
`epoch_len(d)` is the epoch length in DAA seconds in force on day `d` (spec 01 section 1.12: day `d` covers `[86,400 d, 86,400 (d + 1))`). Epoch `(d, e)` covers DAA scores `[86,400 d + epoch_len(d) e, 86,400 d + epoch_len(d) (e + 1))` for `e` in `0 .. 86,400 / epoch_len(d)`. Every ladder step divides 86,400, so the last epoch of a day ends exactly at the day boundary and the first epoch of a day starts at it, as 1.12's day key already assumes ("the program seed of the first epoch of day d").
The identifier of an epoch is its start score `s = 86,400 d + epoch_len(d) e`, not an index. Everything that today takes the epoch index `e` (the seed checkpoint rule of spec 04 section 4.3 step 1, `seed_source` in the header, the hot table key of layer 5, the program id) takes `s` instead. At the base length `s = 3,600 e`, so nothing changes for the devnet.
"Which program was this block mined under" stays a function of the header alone: the header's DAA score gives `d`; `epoch_len(d)` is a function of the chain state before day `d - 2` (section 2.3), which is in the header's own past, exactly as the era parameters `E_n` are; so `s` follows, then `C(s)`, then the seed. Two nodes validating the same header derive the same `s`.
### 2.2 Why the day and not the era
Anchoring to the era would make a change wait up to 180 days, which defeats the purpose ("miners can shorten it when an FPGA appears"). Anchoring to the day makes the response time the signalling window plus two days. The cost is one more thing the day boundary does; it already resets the day key, the cache and the dataset, so the program change at a day boundary is already paid. The parameter still lives in the era table of 1.13.1 (its base, bound and draw slot are genesis constants there); only its activation clock is the day.
### 2.3 The signal
Spec 05 section 5.7's mechanism, with its open items (window length, delay, field) answered here for this parameter only:
| Item | Value | Note |
|---|---|---|
| Field | 3 bits of the header `version` (Kaspa's version-bits field), the ladder index 0 to 7; 0 to 7 all valid, the current index is the default a miner carries | the proposal-id format of O-5.3 stays open for code upgrades; this is a parameter vote, not a code upgrade |
| Window | 7 days of blue blocks (604,800 at 1 block/s, approximate: counted in blocks, not time) | long enough that a rental burst cannot swing it; short enough to answer an FPGA in a week |
| Threshold | 90% of blue blocks in the window carry the same index, and it differs from the index in force | the 90% rule of 5.7; a split vote changes nothing |
| Activation | the first day boundary at least 2 days after the window closes | 2 days covers the 1,200-s seed lead and every compile-ahead measured in section 6 many times over; a node sees the change coming 2 days ahead |
| Hysteresis | none beyond the 90%: to move again, 90% must carry a new index in a fresh window | |
`epoch_len(d)` is therefore: the base (3,600) at genesis; after each activation, the activated index's length. A node derives it from the blue blocks of the window, which are in the header's past.
### 2.4 The draw slot
The era stream of `era-layout.md` section 1.1 draws seven values in a fixed order. This parameter takes draw 8: `next()` is consumed and the value is not used. Reason to consume it: the stream layout is fixed at genesis; if a draw is ever wanted for this parameter it changes no other parameter's value. Reason not to use it: a drawn length answers no threat (an FPGA fleet is defeated by the minimum, not by the variance), and it would move the difficulty-settle share (section 4) and the VDF duty (section 3.3) at random every 180 days.
### 2.5 The genesis reserve row (spec 01 section 1.13.1)
| Parameter | Base (era 0) | Draw | Bound |
|---|---|---|---|
| Epoch length `epoch_len` | 3,600 DAA s | draw 8 of the era stream consumed, not used; set by 90% signal (section 2.3) at a day boundary | the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 |
Unlock condition: live at genesis at the base; the signal path is the unlock. Nothing moves without 90% of blue blocks over 7 days. No height gate is needed because the base is today's value.
## 3. The seed path at a shorter epoch
### 3.1 Two options for the VDF
Spec 04 section 4.3: the seed of epoch `s` is the 10-minute VDF (`T_epoch`, 600 s on the reference core) of the checkpoint at least 1,200 DAA s before `s`. The output is known to a reference core 600 s before the epoch starts (a core half as fast finishes at the boundary). So the usable compile-ahead window is 600 s, at every length.
| Option | Rule | Known-program window | Grinding margin (spec 04 section 4.6, against the 2-s window) | Checkpoint depth | VDF duty per core | Verdict |
|---|---|---|---|---|---|---|
| A | `T_epoch` and the 1,200-s lead stay genesis constants whatever `epoch_len` | 600 s at every length (lead minus evaluation) | 300x at every length | 1,200 s plus d at every length: a reorg across it stays a merge-depth-scale event (merge depth 3,600 s, spec 04 section 4.3) | `600 / epoch_len`: 17% at 3,600, 33% at 1,800, 100% at 600 | RECOMMENDED |
| B | `T = epoch_len / 6`, lead `= epoch_len / 3` | `epoch_len / 6`: 100 s at 600 | `epoch_len / 6 / 2 s`: 50x at 600 | `epoch_len / 3`: 200 s at 600, a tenth of the merge depth, so a reorg across the seed checkpoint is an ordinary event | 17% at every length | not recommended: the seed checkpoint gets shallow and the margin falls 6x at the floor |
A at the floor: the checkpoint for epoch `s` is at `s - 1,200`, two epochs back; the VDF finishes at `s - 600`, the start of the previous epoch; so each epoch's program is known for exactly one epoch before it starts. "The seed input for `s + 1,200` is known during epoch `s`" is true but useless to a grinder or a compiler: the input does not give the program before the VDF finishes.
### 3.2 Grinding (proto-vdf/README.md, grinding table)
The table's model is per epoch: a favourable program is worth `3,600 s (1 + a) / (1 + s a)` blocks of revenue for the epoch, against one burned block. The benefit is proportional to the epoch length and the cost is not, so without the delay the gain-to-cost ratio falls in proportion: 6.0 to 15.1 : 1 at 3,600 becomes 1.0 to 2.5 : 1 at 600 (the same model, scaled; not re-run) and 12 to 30 : 1 at 7,200. With the delay the gain is 0 at every length, because the grinder learns nothing inside the 2-s window; option A keeps the 300x margin that makes that true. A shorter epoch weakens the attacker's prize and leaves the defence as it is; a longer one raises the prize and the defence still holds at 300x.
### 3.3 The VDF duty (a consequence, every tier)
Under A a node evaluates a 600-s VDF once per epoch, so one core is busy `600 / epoch_len` of the time: 8% at 7,200, 17% at 3,600, 33% at 1,800, 50% at 1,200, 100% at 600. At the floor every mining node spends one core on the VDF without pause (the M5 Max core at 163,000 squarings/s, spec 04 section 4.2; x86 with AVX-512 IFMA faster, open item there). What that means per tier: a pool user, nothing (the pool evaluates); a home miner on a 4-core box, a quarter of the CPU at the floor and the option of spec 04 section 4.7 B (take `(y, pi)` from a peer, verify in 4.5 ms) if it has no core to spare; a rig, one core for all its cards. Proof traffic: 516 bytes per epoch, 6x per hour at the floor (3 KB an hour); verification 4.5 ms per epoch. The floor is an emergency setting for the day an FPGA appears; at the base nothing changes.
### 3.4 Fallback (spec 04 section 4.7)
Unchanged. Option B there (no consensus fallback, proofs from peers) matters more at the floor, because a node slower than 2x the reference core and isolated loses up to a whole epoch instead of a sixth of one. Decision at gate 3 as written.
## 4. The difficulty window as a constraint on the floor
Spec 02: rule v2's reference lane is the newest `REF_WINDOW_V2 = 600` DAA s of the epoch, the short lane is 120 chain blocks, and the lanes are epoch-bounded because the hash rate steps by program (35 to 48 Mhash/s across seeds on the M5 Max, spec 01 section 1.12). The constraint:
| Quantity | Rule | At 600 | At 1,800 | At 3,600 | At 7,200 |
|---|---|---|---|---|---|
| `REF_WINDOW_V2` | `min(600, epoch_len)`; the lane never spans a program change | 600 (the whole epoch) | 600 | 600 | 600 |
| Settle after an epoch step, measured | 144 s on the +-30% epoch-step scenario (simulator, `docs/bench-log.md` 3 October difficulty entry; the live v2 settled a 1.85x in-epoch step within 10 minutes, `docs/analysis/difficulty-2026-10-04-oscillation.md`) | 24% of the epoch | 8% | 4% | 2% |
| Rate error during the settle | bounded by the step (the program-to-program spread, +-15% on the M5 Max; the sign is random per program) | | | | |
At the floor the reference lane is the whole epoch and the v2 cap does nothing; the short lane tracks, as the measurement says it does within 10 minutes. The cost is the settle share: with the measured 144 s a 600-s epoch spends about a quarter of its blocks re-settling after each program change, at a rate error up to the program step. Emission follows the block rate (spec 02: `E` per block), so emission wobbles by up to the step for 144 s per boundary; the sign is random per program, so the drift averages near zero and the variance rises. The 144 s is the +-30% scenario; the +-15% program step is not measured (owed, section 9: `sim/difficulty/sim.py` has `EPOCH = 3600` as a constant). Under the under-10% rule applied to this cost the number lands at 1,800 s, which is why the ladder has steps between the floor and the base: the signal can stop at 1,800 and the floor stays reserved for the case that needs it.
## 5. The threat it answers, and the one it does not
### 5.1 An FPGA fleet with a hard datapath per program
A program is 64 instructions x 8 iterations with 16 loads per iteration from a 1 GiB table (spec 01). A fleet that maps each hour's program to a fixed datapath must synthesise, place and route a bitstream per program and load it, inside the 600-s window of section 3.1. Published compile times:
| Source | Device | Figure |
|---|---|---|
| Xiao et al., "Reducing FPGA Compile Time with Separate Compilation for FPGA Building Blocks" (PRflow), FPT 2019, University of Pennsylvania, https://ic.ese.upenn.edu/pdf/prflow_fpt2019.pdf | ZCU102 (XCZU9EG, about 600 k logic cells, approximate) | monolithic Vivado compile of their benchmarks 42 minutes typical, one case 160 minutes; "hour-long compilation times"; their partitioned flow 12 to 18 minutes |
| Aldec, "Save hours of Place & Route time", https://www.aldec.com/en/company/blog/92--save-hours-of-place-and-route-time-in-seconds | Virtex UltraScale class (4.4 M logic cells) | place and route in hours for large designs; UG904's incremental flow about 3x faster when at least 95% of cells and nets are unchanged |
| AMD UG904, Vivado Implementation User Guide (cited through the Aldec blog and the Vivado documentation) | | place and route is the longest stage; incremental reuse needs a 95% similar design, which a fresh random program is not |
The window is 600 s at every ladder step (section 3.1); no row above fits it, so a per-program bitstream can never mine the start of an epoch. What the epoch length sets is the share of each epoch the fleet mines once the bitstream lands: `max(0, 1 - (compile - 600) / epoch_len)`.
| Compile time per program | Share of each epoch mined at 600 | at 1,800 | at 3,600 | at 7,200 |
|---|---|---|---|---|
| 12 min (PRflow partitioned, a small kernel) | 0% | 93% | 97% | 98% |
| 42 min (PRflow monolithic) | 0% | 0% | 47% | 74% |
| 160 min (PRflow worst) | 0% | 0% | 0% | 0% |
| hours (Aldec, large parts) | 0% | 0% | 0% | 0% |
Reading: at the base an FPGA fleet with a fast compile farm mines half to all of each hour; at 7,200 a small kernel is a non-issue for it; at 600 nothing it compiles ever runs. That is the lever the signal gives miners. A compile farm does not help a fleet: the window is wall time per program, not throughput, and the same window applies to every board.
### 5.2 Partial reconfiguration
Loading a partial bitstream is fast: the ICAP takes 4 bytes per cycle at 100 MHz, 400 MB/s, so a region reloads in milliseconds (Microsoft Research, "Minimizing Partial Reconfiguration Overhead with Fully Streaming DMA Engines and Intelligent ICAP Controller", measured 399.6 MB/s, https://www.microsoft.com/en-us/research/wp-content/uploads/2016/02/Minimizing20Partial20Reconfiguration20Overhead20with20Fully20Streaming20DMA20Engines20and20Intelligent20ICAP20Controller.pdf; AMD UG909, Dynamic Function eXchange, https://www.amd.com/content/dam/xilinx/support/documents/sw_manuals/xilinx2021_1/ug909-vivado-partial-reconfiguration.pdf). It does not shorten the compile: the partial bitstream is placed and routed by the same flow first. A smaller region compiles faster than the whole part (the PRflow rows above are that effect), which is why the 12-minute row is in the table; the 600-s window still beats it.
### 5.3 What the layer does not answer
A programmable chip: a soft processor or a GPU-like overlay on an FPGA, or a custom chip with a programmable core, runs any program with no synthesis, and the epoch length does nothing to it. Those are answered by the other layers (the latency bound, the cache and dataset sizes, the mixer, the hot table, the era draws of layers 4 and 8), and an overlay pays area and clock against hard logic (not quantified here; approximate). The public claim does not change by this layer: it removes one route (per-program bitstreams), it does not lower the chip model's headline row.
## 6. The measurement: compile-ahead per card, and the floor
What a card does between receiving the next seed and swapping (`app/igneum-app/src/engine.rs` `prepare_worker`: export the pack from the node; the worker's `prepare`: generate and compile the program, fill the cache, build the dataset, self-test, optionally race the variants, then swap at the boundary): the per-epoch part is program generation plus kernel compile, the hot table fill (layer 5, per epoch) and the variant race where it runs. The dataset is per day, excluded here; today's worker rebuilds it at every prepare because the pack couples the program to the day, which is a worker change owed before the ladder goes below the base (section 9).
### 6.1 Per card, measured or cited
| Card, compiler | Program (generate + compile) | Hot table fill (layer 5, `docs/plans/hot-table.md`, ca2-cache) | Variant race | Compile-ahead total, race off | Total, race on | Source |
|---|---|---|---|---|---|---|
| Apple M5 Max, Metal | 82 ms (hot-swap entry, 4 October); 58 ms (serve check); 0 to 444 ms over 8 live boundaries (M11 fleet row); 1,798 ms with the Metal compiler cold (variant-racing entry); tonight: 15.9 / 17.7 / 20.4 ms min / median / max over 10 fresh programs, 79 ms for the devnet pack's two libraries (6.2) | 0.07 to 0.22 ms on the GPU | 34.0 / 34.9 / 37.8 s min / median / max over 8 boundaries (M11); 39.8 s one round in the serve check | 0.5 s (1.8 s cold) | 38 s | `docs/bench-log.md`: "first hourly program swap on the live devnet" (4 October), "miner performance: variant racing" (4 October), M11 table (5 October); section 6.2 |
| NVIDIA RTX 5090, NVRTC 12.8 | NVRTC 151 to 180 ms (M11; 151 ms at the PC 2 14:20 boundary); prepare total 580 to 1,074 ms including cache 68, dataset 113 and self-test 511 ms, so about 0.5 to 1.0 s without the dataset; 1,285 ms on the nvcc path of 4 October | 0.67 ms per 256 MiB cache fill on the 5090 scaled to `S / 256`, under 1 ms | 232 to 300 ms to compile 17 variants, 112 s of timing for 3 rounds (M11 race rows, PC 1); one round about 37 s (112 / 3, approximate) | 1.0 s | 38 s | `docs/bench-log.md` M11 table and race rows (5 October), hot-swap entry (4 October), "PC 2 at the 14:20 boundary" (4 October) |
| AMD RX 9070 XT, OpenCL (gfx1201, Adrenalin 26.9.2) | NOT MEASURED at the current worker: `proto-opencl/host.c` times `clBuildProgram` only in the `prepare` path (`buildMs` in the `prepared` line) and no `prepared` line from this card is in any upload (PC 1 app logs 17:26, 18:10, 19:02 UTC; jobs `rdna4-serve-4`, `rdna4-bench-1`, `run-readwidth-9070-20261005c` print cache, dataset and check only) | not measured on AMD | none: the OpenCL worker has no race (`docs/design/miner-tuning.md`: 17 names on NVIDIA, 14 on Metal) | 0.31 s plus the compile (cache 9 ms, self-test 300 ms measured, `rdna4-serve-4`) | the same | OWED (section 9) |
| AMD Radeon integrated gfx1036 (PC 2), OpenCL | prepare total 6.9 / 9.4 / 11.7 s with the 1 GiB dataset build on the iGPU inside; the compile is not separated | | none | under 11.7 s | the same | M11 table; the compile share OWED |
| Intel UHD (US laptop), OpenCL | OpenCL build 3.0 to 6.4 s; prepare total 7.3 / 7.9 / 11.7 s | | none | 6.4 s | the same | M11 table |
| AMD Radeon integrated gfx1036 (PC 1), OpenCL, beside WSL build jobs | prepare total 55.3 / 115.8 / 123.8 s (the dataset build on a loaded iGPU); 2 of 10 boundaries compiled inline | | none | the outlier: 124 s with the dataset | | M11 table |
### 6.2 The Mac, tonight (Measured)
Apple M5 Max (Darwin 25.6.0, 64 GiB), 5 October 2026 21:18 UTC, load average 11 to 14 (other agents' builds and runs; the measure lock held for the 3-s run, `with-lock.sh measure bash scratchpad/epoch-measure.sh`), `proto-metal/igneum-bench` built from this branch (e752fc7 + this document) with `swiftc -O -target arm64-apple-macos11`. Ten distinct programs (seed strings `igneum-devnet-v4-epoch0`, `/epoch1` .. `/epoch9`, version 2 generator, 128 loads per hash), each generated and compiled at run time (`makeLibrary` from source plus `makeComputePipelineState`), dataset 2^28 words, one 2^20 batch and one verify warp per program so the run is the compiles:
./igneum-bench --seed igneum-devnet-v4-epoch0 --hours 10 --dataset-log2 28 --batch-log2 20 --batches 1 --verify-warps 1
| Compile (library + pipeline), ms | Values over the 10 programs |
|---|---|
| each | 18.8, 17.8, 17.6, 16.1, 18.6, 15.9, 17.6, 17.6, 20.4, 18.2 |
| min / median / max | 15.9 / 17.7 / 20.4 |
| library share | 7.7 to 9.8 ms; pipeline 8.2 to 10.6 ms |
| hash rate during the run (GPU time, loaded Mac) | 27.2 to 30.1 MH/s, all 10 verify warps PASS |
The same pack three times through `packbench --pack ../proto-cuda/packs/igneum-devnet-v4-epoch0 --batches 1 --batch-log2 20 --group 256` (two libraries, `memhard.metal` and `program.metal`): compile 79 ms, then 1 ms and 1 ms, the system shader cache answering the identical source; cache fill 0.6 to 0.7 ms GPU, dataset build 20.7 to 20.8 ms GPU for 1 GiB.
Reading: a fresh program costs this card about 18 ms to compile with the Metal compiler service warm, 79 ms for a pack with the dataset kernels, and up to 1.8 s when the compiler is cold (the variant-racing entry's first seed). The fleet's 0 to 444 ms per boundary (M11) sits between those, so the live figure is the app's cold-start states, not the compile itself. None of it is visible against a 600-s window: the Mac's compile-ahead is the race or nothing.
### 6.3 Shares and the floor
The usable window is 600 s on the chain (section 3.1: the VDF output lands 600 s before the epoch on a reference core) and 600 DAA s on the devnet (the stand-in lead; the miner sends `prepare` at about 449 DAA before the boundary, confirm = lead / 4, hot-swap entry).
| Card | Compile-ahead (s) | Share of 600-s epoch | 1,800 | 3,600 | 7,200 | Share of the 600-s window | Fits the devnet's 449-DAA prepare point |
|---|---|---|---|---|---|---|---|
| M5 Max, race off | 0.5 (1.8 cold) | 0.1% (0.3%) | 0.03% | 0.01% | 0.01% | 0.1% | yes |
| M5 Max, race on | 38 | 6.3% | 2.1% | 1.1% | 0.5% | 6.3% | yes |
| RTX 5090, race off | 1.0 | 0.2% | 0.06% | 0.03% | 0.01% | 0.2% | yes |
| RTX 5090, race on (one round) | 38 | 6.3% | 2.1% | 1.1% | 0.5% | 6.3% | yes |
| RX 9070 XT | 0.31 + compile (owed) | | | | | | |
| Intel UHD | 6.4 (11.7 with the dataset) | 1.1% (2.0%) | 0.4% | 0.2% | 0.1% | 1.1% | yes |
| Radeon integrated, PC 2 | under 11.7 with the dataset | 2.0% | 0.7% | 0.3% | 0.2% | 2.0% | yes |
| Radeon integrated, PC 1 under load | 124 with the dataset | 20.7% | 6.9% | 3.4% | 1.7% | 20.7% | yes, 449 s (but it already compiled inline twice at the base) |
The floor by the rule (the slowest card's compile-ahead under 10% of the epoch and inside the window), dataset excluded: 600 DAA s. The slowest measured compile-ahead is the variant race at about 38 s on both the Mac and the 5090, 6.3% of a 600-s epoch and inside the window. Two conditions carry it:
1. The race, if it stays on, costs 6.3% of mining time at the floor against 1.1% at the base. The M11 race rows already found that base wins on both the 5090 and the Mac with the GPU to itself, so the race should default off (or run once a day), which takes the slowest row to 1.0 s. With the race off the floor is set by the Intel iGPU's 6.4-s OpenCL build at 1.1% of 600 s, with the 9070 XT owed.
2. The iGPU tier under load (PC 1's 124 s) fails the 10% rule below 1,240 s because it rebuilds the dataset per prepare. The per-day dataset reuse (section 9) removes that; until it ships, that tier compiles inline at the floor and loses the first minutes of each epoch, as it already did twice at the base.
## 7. Consequences per tier (the standing rule)
| Tier | At the base (3,600) | At the floor (600) | What to do |
|---|---|---|---|
| Home miner, one NVIDIA card, 8 to 32 GB, Windows or Linux | 1 s per hour of prepare, unchanged | 1 s per 10 minutes: 0.2% | nothing; memory unchanged (this parameter adds no bytes) |
| Home miner, one Apple card (M-series), macOS | 0.5 s plus the race's 35 s per hour | race off: 0.5 s per 10 minutes; race on: 6.3% of mining | ship the race default off (M11 finding) |
| Home miner, one AMD card (RDNA 4), Windows or Linux | compile not measured | the same, owed | measure `prepared` on the 9070 XT (section 9) |
| Integrated GPU (AMD or Intel iGPU) | 7 to 12 s per hour, 55 to 124 s under CPU load | 2% of the epoch, or 21% under load with the per-prepare dataset build | per-day dataset reuse in the worker before any signal below the base |
| Rig (several cards, one node; the app) | one pack export per epoch under `EXPORT_LOCK`, cards prepare in parallel | 6x the exports per hour, serialised, each sub-second | nothing |
| Rig on the installer scripts (`packaging/hive/h-run.sh` step 4, `packaging/linux/bin/igneum-miner.sh` step 4, branch `rig-install`) | the miner runs with `--prepare-packs` and `--exit-on-seed-change`: a worker that prepares swaps in place (the hot-swap entry: `rebuilds 0`); exit 42 is the fallback when a prepare misses, and it restarts the miner and re-exports the pack (the loaded iGPU missed 2 of 10 boundaries at the base, M11) | a miss costs a restart plus a pack export plus the inline compile per 10 minutes instead of per hour; a worker with no prepare support (the built-from-source launcher path, `engine.rs` line 2046) restarts at every boundary, six times an hour | the rig installer (a3e7b2b03222f5cff): prepare-ahead must be the only boundary path on the rig before any signal below the base; exit 42 stays as the safety net but a miss at the floor is a 10%-of-epoch loss, so the miss causes (the per-prepare dataset build on a loaded iGPU, the prepare sent late) are fixed first (section 9) |
| Pool user | the pool compiles | the same | nothing |
| Every node's CPU | one core 17% busy on the VDF | one core 100% busy | section 3.3; peers' proofs for a node with no core to spare |
| The chain | difficulty settles 144 s per hour (4%) | 24% of blocks in settle, rate error up to the program step | the ladder's middle steps (1,800: 8%) before the floor; the +-15% settle measurement owed |
## 8. Spec text
### 8.1 Section 1.12, the epoch row and the paragraph after the table (replacing "Epoch `e` covers DAA scores ...")
| Clock | Length | What changes | Label |
|---|---|---|---|
| Epoch | `epoch_len(d)` DAA s, base 3,600; the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200; set by 90% miner signal at a day boundary (section 1.13.1, section 5.7) | The program: new seed words from the VDF of section 4, new kernel | Designed (Counter ASIC 2.0, layer 9, `docs/plans/epoch-length.md`); 3,600 stays the prototype value on every network until a signal moves it |
Epoch `(d, e)` covers DAA scores `[86,400 d + L e, 86,400 d + L (e + 1))` with `L = epoch_len(d)` and `e` in `0 .. 86,400 / L`; every ladder step divides 86,400, so day boundaries are epoch boundaries. The epoch is identified by its start score `s = 86,400 d + L e`. The epoch of a block is the epoch of its own DAA score, and `epoch_len(d)` is a function of the blue blocks of the signalling window that closed at least 2 days before day `d` (section 1.13.1), which are in the header's past; so "which program was this block mined under" is a function of the header alone once the seed is known. The program for epoch `s` is `generate_from_seed_bytes(program_seed_s)` as before, with `program_seed_s` the VDF output of section 4.3 for the checkpoint at least 1,200 DAA s before `s`. `T_epoch` and the 1,200-s lead are genesis constants and do not follow `epoch_len`: the program of every epoch is known 600 s before it starts on the reference core, at every length. At the base `s = 3,600 e` and nothing here differs from the previous text.
### 8.2 Section 1.13.1, one row added to the parameter table, and one paragraph
| Parameter | Base (era 0) | Draw | Bound |
|---|---|---|---|
| Epoch length `epoch_len` | 3,600 DAA s | one draw of the era stream consumed and not used (the value is set by signal) | the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 |
`epoch_len` is the one era-table parameter set by miners rather than by the draw: 90% of blue blocks over a 7-day window carrying the same ladder index (3 bits of the header version) sets that length from the first day boundary at least 2 days after the window closes (section 5.7). It is not a code upgrade: the rule, the ladder and the window are genesis constants, and the chain carries no release. The era stream consumes its draw so that a future draw of this parameter changes no other parameter's value.
### 8.3 Section 4.3, one sentence after step 3
The lead and `T_epoch` are fixed whatever the epoch length of section 1.12: at the floor of 600 DAA s the checkpoint is two epochs back and the program is known one full epoch ahead; at the base it is known for the last sixth of the previous epoch.
## 9. Owed
| Item | What | How |
|---|---|---|
| RX 9070 XT compile time | `clBuildProgram` on gfx1201 at the current worker | a `prepared` line from the card: PC 1 job with `igneum-worker-opencl --serve` on the 9070 XT across one boundary, or `--bench-pack` with a `build` ms print added beside `cache dataset check` in `host.c` (a 2-line change) |
| Radeon integrated compile share | the compile inside the 6.9 to 11.7 s totals | the same print |
| Per-day dataset reuse in the workers | a prepare within the same day keeps the dataset and rebuilds the program only | `proto-metal/main.swift`, `proto-cuda/nvrtc/worker.cpp`, `proto-opencl/host.c`: the pair holds a day key; needed before any signal below the base for the iGPU tier |
| The race default | off, or once a day | the M11 finding; `proto-metal/main.swift` `race = "on"` today |
| The rig scripts' boundary path | prepare-ahead as the only path on a rig; exit 42 kept as the net, never the routine | `packaging/hive/h-run.sh` and `packaging/linux/bin/igneum-miner.sh` (branch `rig-install`, a3e7b2b03222f5cff): a prepare-miss counter in the status line, and the built-from-source launcher path (no prepare support) retired before any signal below the base |
| Difficulty settle at a +-15% step and at 600-s epochs | the share of each epoch in settle at the floor | `sim/difficulty/sim.py` with `EPOCH` as a parameter and the program-step scenario at +-15% |
| Hot table fill on AMD | the per-epoch fill on the 9070 XT | ca2-cache's PC rows |
| The signal's encoding | the 3-bit field in `version` against Kaspa's use of the field | spec 05 section 5.8 |
## 10. Rows for `docs/plans/counter-asic-2.md`
The layer table:
| 9 | Epoch length as a signalled era parameter: base 3,600 DAA s, ladder 600 to 7,200, set by 90% signal at a day boundary, lead and `T_epoch` fixed | a per-program bitstream (an FPGA with a hard datapath): at 600 s nothing it compiles ever runs (42 to 160 min per compile, PRflow FPT 2019; hours on large parts, Aldec) | compile-ahead 1 s per epoch on the 5090, 0.5 s on the M5 Max (38 s with the race on); one CPU core `600 / epoch_len` busy on the VDF | reserve-only tonight: `docs/plans/epoch-length.md` |
The level 3 numbers row:
| Epoch length | 3,600 DAA s at launch; miners can signal it down to 600 (an FPGA defence, no fork) | floor 600: the slowest compile-ahead (the variant race, 38 s) is 6.3% of the epoch and inside the 600-s seed window; at 600 a per-program FPGA bitstream mines 0% of each epoch, at 3,600 up to 47% (42-min compile) | Measured (Mac compile, 5 October), cited (5090, FPGA compile times), owed (9070 XT compile) |

177
docs/plans/era-layout.md Normal file
View file

@ -0,0 +1,177 @@
# Era layout: table layout and working set drawn per era and per program (Counter ASIC 2.0, layers 4 and 8)
5 October 2026, branch `ca2-era`, worker "ca2-era". Status: Designed and Implemented behind the class flag (`LoadClass::era`, not the lottery hash); nothing here changes the default generator, the pinned packs or any live program. Measured sections are marked as such; everything else is design.
Plan: `docs/plans/counter-asic-2.md`, layers 4 ("table layout drawn per era: item size, stride, interleave") and 8 ("working-set size drawn per program"). The era seed is `E_n` of spec 04 section 4.4. Confirmed on 5 October 2026 by grep over `igneum-pow/src` on branches master, readwidth and opencl-rdna4-telemetry: no era draw existed in code before this branch (`igneum-era`, `EraParams`, `era_seed`: no match).
## 1. What is drawn, and from what
Two streams, nothing else:
| Stream | Seeded from | Draws | Sets |
|---|---|---|---|
| Era stream | `seed_words_from_bytes("igneum-era/" \|\| E_n)` words 0 and 1 (spec 01 section 1.13.1 wrote `n_le64 \|\| E_n`; the index is dropped here because `E_n` already commits to `n` through the VDF input of section 4.4 step 2, and the node's seam hands the generator the era bytes alone: `Epoch::from_chain_seeds(epoch, day, era, class, label)`, branch ca2-v3) | 7 per era, fixed | the load width `W`, the stride `(M, R)`, the interleave `pos[0..3]` |
| Program stream | the epoch seed words as today (spec 01 section 1.3.3) | 11 per instruction instead of 9: the 9 of version 2, then 2 window draws (a class whose loads are not version 2's takes the read-width width roll between them, ca2-v3's rule) | per load site: the window shrink `k_off` and its offset `o` |
The era parameters change the class of every program of the era. The program stream does not see the era parameters (two eras with the same epoch seed draw the same instruction list and the same windows, and differ in width, stride and layout); this is what makes the six era packs below a controlled comparison.
### 1.1 Era draw (proposed spec text for section 1.13.1, replacing its parameter table)
One SplitMix64 stream `S` seeded with `lo = words[0] | (words[1] << 32)` of `seed_words_from_bytes("igneum-era/" || E_n)`. Seven draws, in this order, whether or not a value is used:
1. `W = allowed[below(|allowed|)]`: the width in words of every dataset load of the era, drawn from the genesis-fixed ascending set `allowed`, a subset of {1, 4, 16} (4, 16 or 64 bytes). A set of one element pins the width; the draw is still consumed. The set is `{1}` (4 bytes, v2's load): the read-width decision of 5 October 2026 (`docs/plans/read-width.md`, "keep v2; w16 the only width that passes the rules and closes nothing") and the adoption rule of the same evening (a draw that changes the bytes per hash changes the rate; the six-era hash-rate spread must stay under 5 percent per card). The set is recorded in every pack (`IGNEUM_ERA_ALLOWED_WIDTHS`, `program.json` `era.allowed_widths`) and enters the program id. The code keeps the draw general so the set can be widened at genesis without a new derivation.
2. `M = low32(next()) OR 1`: the stride multiplier, odd, so `x -> x * M` is a bijection on 32-bit words.
3. `R = 1 + below(31)`: the stride rotation, in 1..31 (never 0: spec 01 section 1.14 item 3).
4. to 7. `r_i = next()` for `i` in 0..3: the interleave draws. Let `b = log2(W)` (0, 2 or 4) and `free = 4 - b`. Let `c = [b, b + 1, ..., 15]` (16 - b candidates). For `i` in `0..free`: `j = i + (r_i mod (16 - b - i))`, swap `c[i]` and `c[j]`. The interleave is `pos = [0, ..., b - 1] ++ sort(c[0..free])`, four ascending bit positions in 0..15. Draws `r_free..r_3` are consumed and ignored.
The era parameters are `(W, M, R, pos)`. Era 0 of the devnet packs is listed in section 5.
### 1.2 Dataset mapping with the interleave (replaces the last sentence of section 1.8.5)
Word `w` of the dataset holds word `j(w)` of item `t(w)`, where `j(w)` is the 4-bit number whose bit `i` is bit `pos[i]` of `w`, and `t(w)` is `w` with bits `pos[0..3]` removed (the remaining bits in order). With `pos = [0, 1, 2, 3]` this is today's `dataset[w] = item(w >> 4)[w AND 15]`, byte for byte.
Properties kept:
- An item has the same value at every dataset size (item derivation is untouched), and because every `pos[i] < 16`, `dataset[w]` is the same at every dataset size of at least 2^16 words. The 1 GiB vectors of an era remain valid for words below 2^28 at any larger size, as today.
- The low `b = log2(W)` positions are 0..b-1, so the `W` words of one aligned load lie in one item (`t` is the same for all of them and `j` runs `j0 .. j0 + W - 1`). The verifier derives one item per lane per load, as today: the 4,096-item bound of section 1.11 holds (16 loads x 8 iterations x 32 lanes, whatever the width).
- The dataset build writes 16 words of one item to 16 addresses `w(t, j)` (a scatter of 4-byte writes instead of one 64-byte line when `pos != [0, 1, 2, 3]`). This is the only GPU cost of the interleave and is paid once per day; section 6 measures it.
What the interleave does and does not buy. A chip that hard-wires today's layout (64-byte items, a 64-byte line per item) reads the wrong 15 words with every word once the era draws another layout; the layout changes every 180 days inside rules fixed at genesis. A chip whose address decoder can permute 28 address lines under firmware control pays nothing for it. The honest claim is the first sentence only. The stride below is the same kind of lever: two integer operations per load on a chip, nothing on a GPU.
### 1.3 Load address (replaces "a load reads one 4-byte word at `src AND MASK`" in section 1.5 for era programs)
For a load site with window draws `(k_off, o)` (section 1.4), a dataset of `2^D` words (`MASK = 2^D - 1`), and register value `x`:
```
k = min(k_off, D - 26) (0 when D <= 26)
y = rotl(x * M, R)
idx = ((y AND (MASK >> k)) OR ((o AND (2^k - 1)) << (D - k))) AND MASK
base = idx AND NOT (W - 1) (W words from base are folded as verify::fold_words)
```
Uniformity: `x * M` with `M` odd and `rotl` are bijections of the 32-bit word, so `y` is uniform when `x` is; `y AND (MASK >> k)` is uniform on the window; the offset picks which of the `2^k` aligned windows. Branch-free, integer only, three operations before the mask (multiply, rotate, and-or) against one today. The emitted text has one form per dialect, checkable by text search (section 1.14 item 2): CUDA and OpenCL `ds[((rotl_imm(rN * 0x........u, Ru) & 0x........u) | 0x........u) & mask]`, Metal the same with `dataset[` and `& MASK]`.
### 1.4 Window draw per load site (layer 8; proposed text for section 1.4.3)
After the nine draws of version 2 (and the width roll, for a class whose loads are not version 2's), every instruction takes two more draws, used only on a load slot:
```
k_off = below(3) window = the dataset, a half or a quarter of it (2^(D - k_off) words)
o = low32(next()) AND (2^k_off - 1) which aligned window
```
Bounds: the window never goes below `2^26` words (256 MiB; `k = min(k_off, D - 26)` in 1.3), which exceeds the largest on-chip cache of any card in the benchmark (the RTX 5090's 96 MiB L2, the RX 9070 XT's 64 MB Infinity Cache, vendor figures), and never above the dataset. At the prototype dataset (2^28) the windows are 1 GiB, 512 MiB and 256 MiB; at the genesis dataset (2^29) 2 GiB, 1 GiB and 512 MiB. The dataset grows by the step schedule recommended to the project lead (spec 01 section 1.13.3 option (b), `docs/analysis/card-lifetime-2026-10-05.md`: power-of-two steps, 4 GiB at year 4, 8 GiB at year 12, 16 GiB at year 28, 32 GiB at year 60, every index `AND MASK`), so the window ceiling follows the steps and the floor stays the genesis constant 2^26 words; nothing in the address of 1.3 needs a range reduction. A program has 16 load sites and so up to 16 windows; the set a program reads is their union (section 7 computes its distribution). The verifier bound is unchanged (1.2).
Why per load site and not per program: a per-program window of a quarter of the dataset would hand a 256 MiB SRAM mirror a third of the hours at the prototype size. Sixteen sites with drawn offsets cover the dataset with high probability (section 7), so the mirror a chip would need is the whole dataset in every hour, and the hour-to-hour variation lands on the memory design (which quarter, which half, how many distinct windows), not on its size.
### 1.5 Program stream (replaces "592 draws per program" in section 1.4.3 for era programs)
16 slot draws, then 64 x 11 = 704: 720 draws per program (768 and 784 for a class with the width roll). On the chain an era program is a class v3 program (branch ca2-v3's seam: `ProgramClass::V3`, generator version 3): its id is `program_id(3, seed words, attempt)` as that branch defines it, and the era it was drawn under is identified beside the id by `IGNEUM_ERA_SEED_HEX` (`packcheck::verify_pack_dir_chain` refuses a pack whose era is not the job's), so the pair (id, era seed) names the program. The experiment classes that are not class v3 (`--class <other> --era ...`) carry the era inside the class id instead: `FNV-1a-64("igneum-program-rw/" || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots || "era/" || allowed[3] || W || M_le32 || R || pos[4])`. The class name is the base name with `-era<first stream word as hex>` (`w4-era401998a5`).
### 1.5.1 The seam (branch ca2-v3, kept as its signatures stand)
`V3_CLASS` is the base class of class v3 (16 loads of 4 bytes, the lottery hash's load; the era rides inside it), `V3_ALLOWED = [1]` its width set. `generate_from_seed_bytes_program_class(label, seed, V3, Some(era))` draws `LoadClass::era(V3_CLASS, era, &V3_ALLOWED)` and stamps generator 3; without era bytes (a template before the era is known) the bare `V3_CLASS` stands. `Epoch::chain_dataset(day, class)` is unchanged: the layout of 1.2 is a property of the program (`program.class.layout()`), applied by the interpreter and by `Epoch::dataset_word`, so the day's cache is shared by every era of a day and the engine keys its caches on `(day, class)` as before.
### 1.6 Acceptance
The rule of section 1.4.6 is unchanged in its tests. Its interpreter mirrors 1.3 at the rule's constant `D = 28` (`idx` as above with `MASK = 0x0fffffff`), as `igneum-pow/src/accept.rs` does. The distinct-address bound counts dataset loads as the read-width branch defines it.
## 2. The era seed on the devnet (stand-in for `E_n`)
Until the 1-hour VDF of section 4.4 is in the node, in the shape of the epoch seed's stand-in (`docs/fork-divergence.md` "Epoch seed"):
| Era | `E_n` |
|---|---|
| 0 | the genesis block hash (32 bytes) |
| n >= 1 | the hash of the last selected-chain block whose DAA score is below `15,552,000 n - 7,200` (the 2-hour lead of section 4.4 step 1) |
The era of a block is `floor(DAA score / 15,552,000)`, a function of the header alone. `E_n` for `n >= 1` is known 7,200 DAA seconds before the era starts, which covers the 1-hour VDF when it arrives and the kernel compile and dataset rebuild now. Test seeds for packs and tests: `E_n` = the 32 bytes (little-endian words) of `seed_words_from_bytes("igneum-era-test/<n>")` (`igneum-pow ... --era igneum-era-test/<n>`, the number only names the pack); raw bytes with `--era <n>:<64 hex>`.
## 3. Memory budget (the project lead, 5 October 2026: under 6 GB on an 8 GB card)
The era layout adds no resident memory: a window is a mask and an offset in the kernel text, the interleave is address arithmetic, the stride is two operations. The whole working set on a card, every item from this branch and the others:
| Item | Bytes | Who |
|---|---|---|
| Dataset (prototype) | 1 GiB | existing |
| Cache (256 MiB, resident only while the day's dataset is built, then free; the decided reading, confirmed by the coordinator on 5 October 2026, `docs/analysis/card-lifetime-2026-10-05.md` carries the per-tier working set with the cache freed as the best case) | 256 MiB peak | existing |
| Layer 5 hot table | the ca2-cache worker's figure | not this branch |
| Scratch per resident warp (read-width variant 5, 32 or 128 KiB per warp) | 2,048 warps x 128 KiB = 256 MiB at most | readwidth branch |
| Output and read-back buffers | 2^24 nonces x 8 B = 128 MiB per dispatch | existing harness |
| Era windows, stride, interleave | 0 | this branch |
The era window never exceeds the dataset, so it never grows the footprint.
## 4. Implementation (behind the flag)
| Piece | Where | What |
|---|---|---|
| `EraParams`, `era_draw`, `LoadClass::era` | `igneum-pow/src/generator.rs` | the 7-draw era stream of 1.1; the class carries `era: Option<EraParams>` beside `mix`, `load_slots`, `scratch`, `scratch_kb`; the two window draws per instruction (`Instr::win`, `Instr::off`); the program id of 1.5 |
| `Layout` | `igneum-pow/src/memhard.rs` | `split(w) -> (t, j)`, `join(t, j) -> w`, `LINEAR = [0, 1, 2, 3]`; `MemhardCpu` and `DatasetSource` carry it |
| `load_index` | `igneum-pow/src/verify.rs` | the address of 1.3, shared by the interpreter and the acceptance mirror |
| Emitters | `igneum-pow/src/emit.rs` | the one load form of 1.3 in Metal, CUDA and OpenCL C; `mh_word`, the three `igneum_build` kernels and the Metal build kernel with `mh_t`, `mh_j`, `mh_addr` when the layout is not linear; `IGNEUM_ERA_*` in `program.h`, an `"era"` object in `program.json` |
| CLI | `igneum-pow/src/main.rs` | `--era <igneum-era-test/n \| n:hex>` and `--era-widths 4,16,64` (one width pins) on every command |
| Tests | `igneum-pow/src/*.rs`, `igneum-pow/tests/packs.rs` | the draw is deterministic and within bounds; `split`/`join` are inverse and the dataset is a prefix at every size; six era programs pass the generator contract and the acceptance rule; vectors round-trip; the pinned packs are byte-identical; the six era packs match the emitters and every load has the form of 1.3 |
The default class is untouched: `LoadClass::V2` has `era: None`, every emitter branch on `era` keeps today's text, and `tests/packs.rs` diffs the pinned packs (`igneum-genesis-mh`, `igneum-devnet-v4-epoch0`) against the emitters as before. Section 6 records the diff of a fresh export against the checked-in files.
## 5. The six era packs
`proto-cuda/packs-ca2-era/era-<n>`, `n` in 0..5: the devnet's 32-byte epoch seed `edc4fa84...fb07` and day bytes `igneum-day/20730` (the seeds of the pinned pack `igneum-devnet-v4-epoch0`, which is the v2 baseline with the same program seed), dataset 2^28 words, era seed `igneum-era-test/<n>`, width pinned at 4 bytes (`--era-widths 4`, the default). The program seed is held fixed so that the six packs differ in the era parameters only (section 1); each carries `seeds.txt` for the one-click workers. Every pack is attempt 1 (attempt 0 of this seed is rejected under the era class: 12 draws per instruction give a different stream from v2's). The drawn parameters (`igneum-pow show --epoch-hex edc4... --era igneum-era-test/<n>`, 5 October 2026):
| Pack | Class | Era seed `E_n` (first 16 hex) | W (bytes) | M | R | pos |
|---|---|---|---|---|---|---|
| era-0 | w4-erab2ed8a89 | 5e0587f455a86e91 | 4 | 0x625e5ab3 | 19 | 0, 2, 10, 15 |
| era-1 | w4-era676a17fc | df57136f2ad5f410 | 4 | 0xb2a9d70d | 6 | 1, 3, 8, 13 |
| era-2 | w4-era843155d7 | 7f450623297a954f | 4 | 0x2b4a5b97 | 28 | 1, 3, 4, 8 |
| era-3 | w4-erad6367bfe | 8bffdd3366b9c3ff | 4 | 0x27ea7eff | 30 | 2, 3, 8, 13 |
| era-4 | w4-era4488f3ed | e593fc1d48475c88 | 4 | 0x4d38603d | 10 | 2, 9, 13, 15 |
| era-5 | w4-eraf897c84e | ff87ad96a1b53f36 | 4 | 0x03ac37ad | 22 | 0, 2, 10, 13 |
All six are class v3 packs (generator 3, `IGNEUM_PROGRAM_CLASS "v3"`, `IGNEUM_ERA_SEED_HEX`), attempt 0, program id `73bcbfe8ccf988f1` in every pack (the seam's `program_id(3, seed, attempt)`; the era seed beside it names the program), the era layout over version 2's item construction (mixer x1, the genesis cache) so that the v2 baseline pack is the same dataset; the chain's class v3 composes the same draw over `LoadClass::MX4` (mixer x4, growth), and the integration re-exports these packs on it after the PC rows. The era-seed-to-pack assignment above is from `program.h` of each pack; the test-seed numbering is only the pack name.
The windows are a property of the program, so they are the same in all six packs (site:shrink:offset): `1:2:3 4:2:2 6:0:0 10:1:0 12:1:1 20:2:0 27:1:0 30:0:0 35:0:0 40:0:0 41:0:0 43:2:3 45:1:0 52:0:0 53:2:2 54:0:0`: 7 sites read the whole dataset, 5 a half, 4 a quarter; the union is the whole dataset.
## 6. Measurements
Pending at the time of this commit; each table below says the machine, the date, the harness and the command when filled.
### 6.1 Byte-identical default path
### 6.2 Bit-exactness on the Mac (Metal, Apple OpenCL, CUDA emulation)
### 6.3 Hash rate per era on the M5 Max (Metal) and the CPU verifier
### 6.4 PCs (RTX 5090 CUDA, RX 9070 XT OpenCL): prepared, waiting for the go
## 7. The chip-model line per draw
From `docs/analysis/m16-recompute-attacker-2026-10-05.md` and the random-read ceilings of `docs/bench-log.md` ("the 9070 XT on the eGPU": 9070 XT 2.42 to 2.68 G loads/s at 1 GiB, 5090 16.4 to 18.0, M5 Max 3.41 to 3.49; every random 4-byte read costs AMD a 64-byte line):
| Quantity | Formula |
|---|---|
| Bytes read per hash | 128 loads x W bytes |
| Distinct 64-byte lines per hash | 128 (one line per load at every W up to 64 bytes; the census's 120 to 128 distinct addresses per hash) |
| SRAM a chip needs to mirror what the hash reads | the union of the program's 16 windows (a distribution over programs; section 7.1) |
| Latency-bound share | measured rate / (the card's 4-byte random-read ceiling / 128) |
The latency-bound share uses the 4-byte ceiling for every width because the 9070 XT line probe showed the same count per second for 4-byte and 64-byte random reads; the 5090's 16-byte and 64-byte ceilings are the read-width branch's measurement, cited when they land.
### 7.1 Union of windows per program
Filled from a CPU census over programs (section 6).
## 7.2 Found on the way (harness defects, both fixed on this branch)
| Where | Defect | Fix | Checked |
|---|---|---|---|
| `proto-cuda/nvrtc/packfile.h` (the one-click workers) | the seed words were re-derived from the bare epoch seed, so every pack of attempt 1 or higher was refused ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT"); 5.14 percent of epochs under v2, all six era packs, and the epoch 34 fleet outage of 18:23Z on 5 October 2026 (branch pack-loop af983a7, which this branch takes: `pf_program_words`) | the pack-loop derivation merged over the readwidth packfile (class fields and string seeds kept) | the devnet pack (attempt 0) loads, the six era packs (attempt 1) load, a copy of era-0 with the attempt tampered to 0 is refused on the re-derivation (section 6) |
| `proto-cuda/host.cu`, `proto-opencl/host.c` | the host-side dataset word was `mh_item(w >> 4)[w AND 15]`, the harness's own copy of the linear layout; under an interleaved layout the "64 random points vs host derivation" check failed while the Mac samples and the vectors passed | `host_ds_word` calls the pack's `mh_word` (memhard.h), which carries the layout | `proto-cuda/emu/test-layout.sh`: the CUDA emulation on era-1 (interleaved) and the devnet pack (linear) must pass the random-point check; it failed on era-1, era-3 and era-5 before the fix (section 6) |
## 8. What is unverified
- Everything in section 6 marked pending.
- The 1-hour VDF does not exist; the devnet stand-in of section 2 is a proposal.
- The interleave's value against a chip with a programmable address decoder is nil (1.2); the claim is limited to hard-wired layouts.
- The window floor of 2^26 words is set by the 5090's L2 (96 MiB) and the 9070 XT's Infinity Cache (64 MB, vendor figures); a future card with a larger cache moves the floor, which is a genesis constant.
- No cryptanalysis of the stride (a multiply and a rotate before the mask); it is a bijection, so the address distribution is that of the register value, as today.

266
docs/plans/hot-table.md Normal file
View file

@ -0,0 +1,266 @@
# Hot table: a second table sized to GPU cache, read beside the 1 GiB dataset
Counter ASIC 2.0, layer 5 (`docs/plans/counter-asic-2.md`). Experiment branch `ca2-cache`, 5 October 2026 (night), on top of the read-width branch (`readwidth` b970dda: `LoadClass`, `verify::fold_words`, the scratch variant, `proto-metal/packbench.swift`, `proto-opencl/host.c --bench-pack`). Nothing here is the lottery hash: every hot class sits behind the generator flag and the default v2 path is byte-identical (the pinned packs `igneum-genesis-mh` and `igneum-devnet-v4-epoch0` are diffed by `igneum-pow/tests/packs.rs`).
## 1. The idea
The honest hash is 128 dependent random 4-byte reads over 1 GiB (spec 01 section 1.5). A chip that wants a gain on that must beat a GPU at DRAM random reads, which is the same physics for both (the plan's last section). What a chip can do that a GPU cannot is choose its memory: a chip can mirror read-only data into SRAM and serve it at SRAM latency, if the data fits.
The hot table turns that around. A second table `H` of `S` MiB (32, 64 or 96 in this experiment) is derived from the epoch seed and read by `k` of the 16 load slots (2, 4 or 8). `S` is chosen to fit the caches of the cards that mine: 96 MiB L2 on the RTX 5090, 64 MB Infinity Cache plus 8 MB L2 on the RX 9070 XT, the system level cache on the M5 Max (figures approximate, from memory; the probe of section 6 measures what each card does at each size). A GPU gets the hot loads as cache hits for free. A chip must carry `S` MiB of SRAM for the same hits, beside the DRAM path it still needs for the other `16 - k` slots, or serve `H` from DRAM and fall behind the GPU by the hot share. Either way the chip pays for something the GPU already has.
What this does not change: the cold loads (the 1 GiB dataset) stay a dependent chain at DRAM latency, so the latency-bound property holds for them; the hot loads are interleaved in the same chain (every load's address is a fresh register of the same iteration, G2), so a hot hit shortens the chain by one DRAM latency and nothing else.
## 2. The specification text (proposed; prototype values, to be fixed at gate 1)
### 2.1 Hot key and fill
For an epoch whose program seed bytes are `e` (spec 01 section 1.12: the UTF-8 of a seed string in the packs, the 32-byte VDF output on the chain; the bytes before the attempt suffix of 1.4.6, so every attempt of one epoch shares one table):
```
KH = seed_words_from_bytes("igneum-hot/" || e) 8 words
```
`H` has `N_H = S x 2^18` words (`S` MiB) in `N_seg = S x 256` segments of 64 chained lines of 16 words, filled exactly as the cache of section 1.8.3 with `KH` in place of `K` and the tag `("Igne", "umHT") = (0x49676e65, 0x756d4854)` in place of `("Igne", "umMH")`:
```
prev = 0^16
for j in 0..63:
in = prev XOR (sigma[0..3] || KH[0..7] || s || j || tag[0..1])
line = B(in) ChaCha12 with feed-forward, section 1.8.2
H[segment s, line j] = line
prev = line
```
One GPU thread per segment, as the cache fill. The fill is a once-per-epoch cost: `S / 256` of the 256 MiB cache fill (section 1.8.3 table: 0.6 to 2.1 ms on the M5 Max GPU, 0.67 ms on the RTX 5090, 175 to 190 ms on one CPU core for 256 MiB), so under 1 ms on a GPU and 22 to 71 ms on one core, measured below.
### 2.2 The hot load
A hot load slot reads `H` instead of the dataset with the same fold (width 1: a plain XOR):
```
hot: dst = dst XOR H[mulhi(src, N_H)]
```
`mulhi(a, b)` is the high 32 bits of the 64-bit product (the `mulhi` family of section 1.4.1, bit-exact on Metal, CUDA and OpenCL). The index lies in `[0, N_H)` for any `N_H`, which is what lets `S = 96` exist: 96 MiB is not a power of two, so `src AND MASK` cannot address it. This is the multiply-shift range reduction section 1.13.3 proposes for the growing dataset, so the hot table is also its first measured use. For `S = 32` and `64` the mapping takes the top 23 or 24 bits of `src` where the dataset load takes the low 28; a fresh source is uniform, so neither choice costs uniformity (section 6, hot-load uniformity test).
The emitted text has one form per dialect, checkable by text search as the mask check of section 1.14 item 2: `hot[mulhi(rN, HOT_WORDS)]` (Metal), `hot[__umulhi(rN, HOT_WORDS)]` (CUDA), `hot[mul_hi(rN, HOT_WORDS)]` (OpenCL), with `HOT_WORDS` a literal of the pack. A hot pack's hash kernels carry exactly `16 - k` masked dataset loads and exactly `k` hot loads (`igneum-pow/tests/packs.rs`).
### 2.3 Which slots are hot
The 16 load slots are drawn first by the partial Fisher-Yates of section 1.4.3, which emits them in a uniformly random order. The first `k` slots in that draw order are the hot slots. Drawn, not fixed, because a fixed pattern (every fourth load, say) would let a pipeline schedule its SRAM reads statically for every hour; drawn costs no extra draw, so a `hot(S, k)` program takes the version 2 stream exactly and is the version 2 program with `k` of its loads redirected (the width roll of the read-width classes is not taken: `LoadClass::takes_width_roll`). A hot slot keeps every rule of a load: the fresh-source draw (G2), the injecting-write count (acceptance (b)), the lane-constant test and the distinct-address count (acceptance (c)); its address is tagged apart from dataset addresses in the count so a hot word and a dataset word at the same index are two addresses.
The class composes with the read-width fields of `LoadClass` (width mix, scratch `k` and `kb`): the hot slots are taken from the drawn slots after the scratch slots, so a class may carry a width mix, a scratch and a hot table at once. The experiment packs below use the version 2 widths and no scratch.
Two forms (coordinator's decision, 5 October 2026, after the first Mac measurement):
| Form | Load slots drawn | Dataset loads per hash | Hot loads per hash | Items per warp (verifier bound) | Class name | What a chip pays |
|---|---|---|---|---|---|---|
| replaced (`hot(S, k)`) | 16, the first `k` in draw order hot | `(16 - k) x 8` | `8k` | `(16 - k) x 256` | `hot64k4` | the on-die-cache recompute chip of `docs/analysis/scratch-soundness.md` 3.4 (the 256 MiB cache in SRAM, items derived on the fly, 333 MH/s at 50 T op/s, 2.4x the 5090) GAINS: a hot load replaces an item derivation (1,170 ops) with an SRAM read, so at `k = 4` its rate rises 1.33x against the GPU's measured 1.05 to 1.22x |
| added (`hot(S, k, added)`) | `16 + k`, the first `k` in draw order hot | `16 x 8 = 128` | `8k` | 4,096, unchanged | `hot64k4a` | `S` MiB of SRAM and `k` reads per iteration for nothing: the 16 item derivations stay; the GPU pays `k` cache hits |
The added form is the one that taxes the named chip; the replaced form stays as the measured record (section 6). In the added form the slot draw is a partial Fisher-Yates of `16 + k` slots over 1..63, so the program stream differs from version 2 (another slot count), and the width roll is still not taken (`takes_width_roll`: the slot count is 16 plus the class's own added hot slots). `loads_per_hash` is `128 + 8k`; the acceptance rule's distinct-address bound scales with it as for the read-width classes.
Program id: `FNV-1a 64 over "igneum-program-rw/" || generator || seed words || attempt || mix || load_slots || "hot/" || S || k [|| "added"]` (the read-width id with the hot fields appended), so no hot pack can be mistaken for a version 2 one, for another hot class or for the other form.
### 2.4 Acceptance (section 1.4.6)
The dynamic test stays a pure function of the program. A hot load reads `dataset_elem(mulhi(src, N_H), SW[2], SW[3])` (the six-operation closed form keyed by seed words 2 and 3, where the dataset stand-in is keyed by words 0 and 1), so the two stand-ins are two tables without any cache or day. Every other test is unchanged; the distinct-address bound is the read-width rule's (dataset and hot loads counted, scratch read-modify-writes not).
### 2.5 The verifier (section 1.11)
A verifier holds, per day, the mixer parameters and the 256 MiB cache, and, per epoch, `H` (`S` MiB, filled on one core in the times of section 6). A hot load is one table read per lane; a dataset load is the item derivation of section 1.11 as before. The verifier never holds the dataset. Per epoch the verifier's memory is `256 MiB + S MiB`.
## 3. Memory budget (the project lead's cap, 5 October 2026: the working set on a card stays under 6 GB on an 8 GB card)
Decided by the coordinator on 5 October 2026 after the card-lifetime review (`docs/analysis/card-lifetime-2026-10-05.md`, branch `card-lifetime` 1fecfe2, merged into `ca2-coord`): the GPU frees the 256 MiB cache after the daily dataset build (the hash never reads it; the rebuild costs 0.67 ms of fill and 13.4 ms of build on the 5090, bench-log 3 October 2026), so the cache is not resident and the working set is dataset + hot table + scratch (0 in v3) + buffers. The table below is that review's per-tier table with the hot table at its largest (96 MiB), buffers 128 MiB (the harnesses' 2^24-nonce output), resident warps = SMs x 48 on NVIDIA Ampere and later (SM counts approximate, from memory; the GTX 1650 is Turing at 32 warps per SM), 2,048 launched warps on Apple (the Metal harness). The scratch columns are the read-width variant's 32 and 128 KiB per resident warp, kept for the record; v3 carries no scratch.
| Tier | Card assumed (SMs, approximate) | Resident warps | Scratch at 32 KiB | Scratch at 128 KiB | Hot table | Dataset at genesis | Buffers | Total, no scratch | Total at 32 KiB | Total at 128 KiB | Years the dataset leaves under mapping (b), cache freed |
|---|---|---|---|---|---|---|---|---|---|---|---|
| 4 GB | GTX 1650 (14 SMs x 32) | 448 | 14 MiB | 56 MiB | 96 MiB | 2,048 MiB | 128 MiB | 2,272 MiB | 2,286 MiB | 2,328 MiB | to year 4 (the 4 GiB step) |
| 8 GB | RTX 3050 (20) | 960 | 30 | 120 | 96 | 2,048 | 128 | 2,272 | 2,302 | 2,392 | to year 12 (the 8 GiB step) |
| 12 GB | RTX 3060 (28) | 1,344 | 42 | 168 | 96 | 2,048 | 128 | 2,272 | 2,314 | 2,440 | to year 28 (the 16 GiB step; year 12 if the cache were resident) |
| 16 GB | RTX 5060 Ti (36) | 1,728 | 54 | 216 | 96 | 2,048 | 128 | 2,272 | 2,326 | 2,488 | to year 28 |
| 24 GB | RTX 4090 (128) | 6,144 | 192 | 768 | 96 | 2,048 | 128 | 2,272 | 2,464 | 3,040 | to year 60 (the 32 GiB step) |
| 32 GB | RTX 5090 (170) | 8,160 | 255 | 1,020 | 96 | 2,048 | 128 | 2,272 | 2,527 | 3,292 | to year 60 |
| Apple 8 to 64 GB | M-series, 2,048 launched | 2,048 | 64 | 256 | 96 | 2,048 | 128 | 2,272 | 2,336 | 2,528 | 8 GB to year 4, 16 GB to year 12, 32 GB to year 28, 64 GB to year 60 (50% of unified memory usable, the review's assumption) |
The prototype packs here use the 1 GiB dataset of spec 1.5 (1,024 MiB less in every total). Every total is under 6 GB with the hot table at its largest, so `S` is not what the cap binds: the dataset's growth is, and the hot table takes 96 MiB of the room at every tier (about 2 months of the 0.5 GiB-a-year schedule). The verifier's memory is `256 MiB + S MiB` (section 2.5); the GPU's is the table above.
## 4. The chip model with the SRAM it would need
Figures: SRAM area per bit from `docs/analysis/m16-recompute-attacker-2026-10-05.md` section 3 (256 MiB in about 100 to 300 mm^2 at a current node: the low end from a 0.02 um^2 bit cell with array overhead, the high end from wafer-scale parts at about 1 MB per mm^2; approximate, from memory) and from `docs/plans/counter-asic-2.md` (256 MB in about 45 mm^2 at a leading node, approximate). That is 0.18, 0.39 and 1.17 mm^2 per MiB. Die areas: a 750 mm^2 GPU-class die (the M16 model's equal-silicon comparison) and a 100 mm^2 memory-chip die (an assumption for a latency-bound chip whose die holds memory controllers and little else; labelled as such).
| S (MiB) | SRAM at 0.18 mm^2/MiB | at 0.39 | at 1.17 | Share of a 100 mm^2 die (0.39) | Share of a 750 mm^2 die (0.39) |
|---|---|---|---|---|---|
| 32 | 6 mm^2 | 12 mm^2 | 37 mm^2 | 11% | 1.6% |
| 64 | 12 mm^2 | 25 mm^2 | 75 mm^2 | 20% | 3.2% |
| 96 | 17 mm^2 | 37 mm^2 | 112 mm^2 | 27% | 4.8% |
The gain arithmetic. Let `G0` be a chip's gain over a GPU on the version 2 hash (the plan's public claim: under 2x; the plan's last section says the bound is DRAM latency, the same physics on both). Let `g` be the GPU's own measured speed-up from the hot class over version 2 on the same card (section 6: `hash rate hot(S, k) / hash rate v2`). The ideal `g` with every hot load a hit at zero cost is `16 / (16 - k)`: 1.14x at `k = 2`, 1.33x at `k = 4`, 2.0x at `k = 8`; the measured `g` says what share of that a real cache delivers while the 1 GiB dataset streams through the same cache.
| Chip | Hot loads served from | Gain after the hot table | Arithmetic |
|---|---|---|---|
| A, no SRAM for H | DRAM | `G0 / g` | the chip's rate is what it was; the honest GPU gained `g` |
| B, S MiB of SRAM for H | SRAM | `G0 x A_die / (A_die + A_S)` per unit of silicon | the chip regains `g` and pays `A_S` on top of its die |
| B on a 100 mm^2 die, S = 64, 0.39 mm^2/MiB | SRAM | `0.80 x G0` | 100 / 125 |
| B on a 750 mm^2 die, S = 96, 0.39 mm^2/MiB | SRAM | `0.95 x G0` | 750 / 787 |
What the big-table loads still cost the chip: `(16 - k) x 8` dependent DRAM reads per hash at the card's loaded latency (the 9070 XT entry's probe: 451 ns on the 5090 at 4,096 lanes, 276 ns unloaded on the 9070 XT, 1,949 ns on the M5 Max at 4,096 lanes; section 6 repeats the probe here). At `k = 4` that is 96 reads per hash; to match one RTX 5090 at its honest 229 Mhash/s (the M16 model's reference) a chip must keep `229 M x 96 x 300 ns = about 6,600` DRAM reads in flight at a 300 ns latency, and 11,000 at 500 ns, whatever its arithmetic. That queue depth is a memory-controller property, which is the latency-bound argument of the plan restated for the cold share.
What the hot table does to the recompute attacker of M16: nothing good. `H` is a plain ChaCha12 chain, so a word of it can be recomputed from `KH` at `j + 1` block evaluations (32.5 on average, about 40,000 integer operations per hot load, approximate, against 1,170 per dataset item), which is 35x the cost of recomputing a dataset word. A chip stores `H` or reads it from DRAM; it does not recompute it.
Reading, before the measurements: the hot table costs a chip `A_S` of die area or `g` of rate. Which one binds depends on the measured `g`, which is why the measurement comes first. If a GPU's cache delivers most of the ideal `g` with the dataset streaming beside it, `k = 8` at `S = 64` doubles the honest rate and halves chip A. If the cache delivers little, the hot table is a cost to the verifier (`S` MiB per epoch) with no gain, and the layer is dropped.
## 5. Implementation (branch `ca2-cache`)
| Where | What |
|---|---|
| `igneum-pow/src/generator.rs` | `LoadClass { hot: Option<HotClass> }` beside `mix`, `load_slots`, `scratch`, `scratch_kb`; `HotClass { mb, k }`; names `hot32k4`, `hot64k2`; `Op::Hot`; the first `k` drawn load slots after the scratch slots are hot; no width roll for a hot class with version 2 widths (`takes_width_roll`); the program id carries `hot/S/k` |
| `igneum-pow/src/memhard.rs` | `HotTable` (key, S, words), `hot_key(seed_bytes)`, `hot_words(mb)`, `hot_segments(mb)`, `hot_index(src, words)`, the tagged segment fill shared with the cache, `HOT_TAG` |
| `igneum-pow/src/verify.rs` | `DatasetSource::hot: Option<HotTable>`; `Op::Hot` in `step`; `Epoch::new_class` and `from_seed_bytes_class` fill `H` from the program's seed bytes when the class has a hot table |
| `igneum-pow/src/accept.rs` | `Op::Hot` with the closed-form stand-in of 2.4 |
| `igneum-pow/src/emit.rs` | the hot load statement in the three dialects; `hot` as the buffer after the init words (Metal buffer 3, or 4 when bound; CUDA and OpenCL argument after `mask`, or after the init words when bound) and before the scratch triple; `HOT_WORDS` literal; `ht_cache_segment` and `igneum_hot_fill` kernels in memhard.h, kernel.cu, kernel.cl and memhard.metal; `program.h` `IGNEUM_HOT_MB`, `IGNEUM_HOT_WORDS`, `IGNEUM_HOT_SEGMENTS`, `IGNEUM_HOT_SLOTS`, `IGNEUM_HOT_KEY_INIT`; `vectors.h` and `vectors.json` the hot head, last line and FNV-1a 64 |
| `igneum-pow/src/main.rs` | `--class hot<S>k<k>` on every command (`bench` reports the fill time and the verifier ms per warp) |
| `igneum-pow/tests/packs.rs` | the five hot packs pinned (program, vectors, every emitted file, the load-form count: `16 - k` masked loads and `k` hot loads) |
| `proto-cuda/packs-ca2-hot/` | replaced: `hot32k4`, `hot64k4`, `hot96k4`, `hot64k2`, `hot64k8`; added: `hot32k4a`, `hot64k4a`, `hot96k4a`; all from seed `igneum-genesis`, day `2026-10-03` |
| `proto-cuda/nvrtc/packfile.h` | `hotMb`, `hotWords`, `hotSegments`, `hotSlots`; the hot self-test values; `pf_selftest` checks them when the pack carries them |
| `proto-opencl/host.c` | `--bench-pack` and the self-test allocate and fill `H` from the pack's `igneum_hot_fill`, check its head, last line and FNV-1a 64, pass it as the argument after the init words |
| `proto-metal/packbench.swift` | the same on Metal from `memhard.metal` |
| `proto-cuda/nvrtc/worker.cpp` | `--check` and `--bench` (new: the base kernel timed over `--batches` dispatches, one RESULT line) allocate, fill and self-test `H`; the launch passes it |
## 6. Measurements
Every row names the machine, the harness and the command; the bench-log entry of 5 October 2026 ("the hot table on the M5 Max") carries the raw lines. The Mac rows were taken on 5 October 2026, 20:19 to 20:21 UTC, under the measure lock, with the Mac's load average at 14 to 27 from other agents' CPU work (the lock serialises builds and measurements, not every process), so they are ordered, repeatable to within a few percent against each other, and not the Mac's quiet numbers. The PC rows wait for the coordinator's go.
### 6.1 Random-read probe at the hot sizes
`igneum-bench-cl-igneum-genesis-mh --memprobe --probe-mib S` (`proto-opencl/host.c`; dependent random 4-byte loads, best of 3, 256 steps per lane, work-group 256; the ceiling is the 4,194,304-lane row; Apple OpenCL reports wall time).
| Card | 32 MiB ceiling | 64 MiB | 96 MiB | 1024 MiB | Ratio 32 / 1024 | 64 / 1024 | 96 / 1024 | ns per dependent load at 4,096 lanes (32, 64, 96, 1024 MiB) |
|---|---|---|---|---|---|---|---|---|
| M5 Max (Apple OpenCL) | 21.7 G loads/s | 12.8 | 12.3 | 3.50 | 6.2 | 3.7 | 3.5 | 1,168; 1,129; 1,242; 1,844 |
| RTX 5090 (CUDA worker `--memprobe`, PC 1, job run-ca2-hot-5090-20261005, card off in the app) | 112.6 | 112.6 | 112.6 | 17.6 | 6.4 | 6.4 | 6.4 | 320; 340; 336; 610 |
| RX 9070 XT (OpenCL worker `--memprobe --device 1`, PC 1, job run-ca2-hot-9070-20261005, card off in the app) | 9.88 (10.8 at 262,144 lanes) | 9.47 | 8.18 | 2.43 | 4.1 | 3.9 | 3.4 | 396; 457; 454; 1,579 |
Reading, RTX 5090: all three sizes sit inside the 96 MiB L2 at one ceiling (112.6 G loads/s, 6.4x the DRAM figure), so the probe alone promises a full hit rate for every S. RX 9070 XT: 32 and 64 MiB inside the Infinity Cache at 9.5 to 10.8 G loads/s (3.9 to 4.1x), 96 MiB at 8.2 (3.4x), as the 9070 XT entry's 64 MiB row said. Streams: 5090 1,563 GB/s at 1024 MiB, 9070 XT 633 GB/s (both at their rated figures, the cards were not parked).
Reading, M5 Max: the step from 32 to 64 MiB halves the ceiling (21.7 to 12.8 G loads/s) and 96 MiB sits with 64, so 32 MiB is inside a cache level that 64 MiB is not (the M5 Max's system level cache size is not published; approximate reading: the 32 MiB table fits, the two larger ones mostly do not and run at a 3.5 to 3.7x advantage over DRAM from whatever hits they get). The coalesced stream grows with the buffer (139, 199, 243, 522 GB/s) because the small buffers are read once from cold.
Predicted `g` from the probe alone, with a hot load costing `1 / ceiling_S` and a cold load `1 / ceiling_1024`: `g = 1 / ((16 - k) / 16 + (k / 16) x ceiling_1024 / ceiling_S)`.
### 6.2 Bit-exactness and hash rate per hot pack
Metal: `proto-metal/packbench --pack <dir> --batches 5 --batch-log2 24` (GPU time). Apple OpenCL: `igneum-bench-cl-igneum-genesis-mh --bench-pack --pack <dir> --batches 5 --batch-log2 24` (wall time). Both harnesses fill the hot table on the device from the pack's `igneum_hot_fill` and check its head, last line and FNV-1a 64 against vectors.json or vectors.h; vectors are the 96 lanes of the Rust reference; the fingerprint is FNV-1a 64 over the 2^24 outputs at base nonce 0.
| Pack | Vectors (Metal, OpenCL) | Hot table FNV (Metal, OpenCL) | Fingerprint 2^24 (both harnesses equal) | Metal Mhash/s | Apple OpenCL Mhash/s | g against v2 (Metal) | Predicted g from the probe | Ideal g |
|---|---|---|---|---|---|---|---|---|
| v2 (igneum-genesis-mh) | 3/3, 96/96 | none | 25f96e7dce90bd4e | 27.68 | 27.61 | 1 | 1 | 1 |
| hot32k4 | 3/3, 96/96 | PASS, PASS | d2e6cf3b61d0b9fe | 33.90 | 33.92 | 1.22 | 1.27 | 1.33 |
| hot64k4 | 3/3, 96/96 | PASS, PASS | e4c5263ac650cc0d | 30.93 | 30.89 | 1.12 | 1.22 | 1.33 |
| hot96k4 | 3/3, 96/96 | PASS, PASS | 5d63439b6e394521 | 29.06 | 28.97 | 1.05 | 1.22 | 1.33 |
| hot64k2 | 3/3, 96/96 | PASS, PASS | 352633bdbbb0d2b6 | 27.67 | 27.27 | 1.00 | 1.10 | 1.14 |
| hot64k8 | 3/3, 96/96 | PASS, PASS | da54630d7dfaaf85 | 47.42 | 46.73 | 1.71 | 1.57 | 2.0 |
| hot32k4a (added) | 3/3, 96/96 | PASS, PASS | 8a3414735db4523c | 25.76 | 25.72 | 0.93 | 0.96 | 1 |
| hot64k4a (added) | 3/3, 96/96 | PASS, PASS | 45668f34105f6307 | 23.92 | 23.87 | 0.87 | 0.94 | 1 |
| hot96k4a (added) | 3/3, 96/96 | PASS, PASS | af763997dfee4c82 | 22.92 | 22.88 | 0.83 | 0.93 | 1 |
For the added form the ideal `g` is 1 (the 16 dataset loads stay) and the probe predicts `g = 1 / (1 + (k / 16) x ceiling_1024 / ceiling_S)`: 0.96 at 32 MiB, 0.94 at 64 and 96 MiB for `k = 4` on the M5 Max; what matters is how far below 1 the GPU lands (its cost of the layer) against the chip's `S` MiB of SRAM and `k` reads.
The PCs (5 October 2026, 21:29 to 21:35 UTC, PC 1 ae432dc7 on app 0.3.9 before and after, the card under test switched off in the app through `api/cards` and restored; CUDA worker `--bench --batches 5 --batch-log2 24 --block-warps 1` on the RTX 5090, wall time around the stream sync; OpenCL worker `--bench-pack --batches 5 --batch-log2 24 --device 1` on the RX 9070 XT, device event time, work-group 256; the v2 references are the readwidth entry's same-night, same-worker numbers: 136.1 and 18.15 MH/s). Every pack bit-exact with the Mac's fingerprint, hot table head, last line and FNV PASS, 96/96 lanes, on both cards.
| Pack | RTX 5090 MH/s | g (v2 136.1) | probe-predicted g (5090) | RX 9070 XT MH/s | g (v2 18.15) | probe-predicted g (9070 XT) | ideal g |
|---|---|---|---|---|---|---|---|
| hot32k4 | 146.6 | 1.08 | 1.27 | 19.79 | 1.09 | 1.23 | 1.33 |
| hot64k4 | 140.8 | 1.03 | 1.27 | 18.73 | 1.03 | 1.23 | 1.33 |
| hot96k4 | 138.5 | 1.02 | 1.27 | 18.33 | 1.01 | 1.21 | 1.33 |
| hot64k2 | 137.5 | 1.01 | 1.13 | 18.17 | 1.00 | 1.11 | 1.14 |
| hot64k8 | 163.6 | 1.20 | 1.73 | 22.32 | 1.23 | 1.59 | 2.0 |
| hot32k4a (added) | 118.7 | 0.87 | 0.96 | 15.27 | 0.84 | 0.94 | 1 |
| hot64k4a (added) | 115.4 | 0.85 | 0.96 | 14.62 | 0.81 | 0.94 | 1 |
| hot96k4a (added) | 114.4 | 0.84 | 0.96 | 14.56 | 0.80 | 0.93 | 1 |
The 5090 at `--block-warps 8` (6 blocks per SM, 8,160 resident warps) is within 0.7% of every row above; the 9070 XT at work-group 32 within 0.5%. Hot fill on the 5090: 1.2 to 2.5 ms (wall, driver API); on the 9070 XT 1.6 to 5.3 ms.
Reading, the PCs. The probe promised a full hit rate for every S on the 5090 (all three tables inside the 96 MiB L2 at one ceiling) and 3.4 to 4.1x on the 9070 XT, and the hash delivered a fraction of it: 1.02 to 1.08x at `k = 4` against the probe's 1.27x and the ideal 1.33x on the 5090, 1.01 to 1.09x on the 9070 XT, 1.20 to 1.23x at `k = 8` against 1.73 and 2.0x. The table that stands alone in the probe does not stand up with the 1 GiB dataset streaming through the same cache: the dataset's random lines evict it (the 5090's L2 and the 9070 XT's Infinity Cache are shared by every load; neither card partitions them). The added form costs the 5090 13 to 16% and the 9070 XT 16 to 20% of its rate for four extra loads per iteration, three to five times the probe's 4 to 7%. The three cards agree on the shape; the 5090's larger cache buys it nothing over the Mac at 32 MiB (1.08 against 1.22).
Hot table fill on the M5 Max GPU (Metal, GPU time): 0.07 ms at 32 MiB, 0.15 ms at 64 MiB, 0.22 ms at 96 MiB.
Reading, M5 Max. Bit-exactness holds: Metal and Apple OpenCL give one fingerprint per pack and every vector lane against the Rust reference, hot table included. On the rate, the hot loads are worth 92% of the probe's prediction at 32 MiB (1.22 against 1.27), 50% at 64 MiB (1.12 against 1.22 is 0.12 of 0.22) and 23% at 96 MiB; at `k = 8` the measured 1.71 is above the prediction (1.57), which says the cold loads also go faster when half of them are gone (fewer dependent DRAM reads in the chain per hash). `hot64k2` gained nothing on this Mac at load (27.67 against 27.68). So on Apple silicon, with the 1 GiB dataset streaming through the same cache, the table that fits (32 MiB) delivers most of its ideal and the larger ones lose most of theirs. The 5090 and 9070 XT, whose caches are the plan's targets, are the PC job.
### 6.3 CPU verifier, one core, avg of 50 warps
`igneum-pow bench --seed igneum-genesis --day 2026-10-03 --class <class> --warps 50` (release build, one M5 Max core, the Mac at load as above; the v2 row from the same session, the 0.604 ms of the brief was an earlier quiet run).
| Class | Hot fill, one core | Items derived per warp | ms per warp | Against v2 |
|---|---|---|---|---|
| v2 | none | 4,096 | 0.626 | 1 |
| hot32k4 | 24.0 ms | 3,072 | 0.489 | 0.78 |
| hot64k4 | 46.4 ms | 3,072 | 0.504 | 0.81 |
| hot96k4 | 73.0 ms | 3,072 | 0.488 | 0.78 |
| hot64k2 | 45.5 ms | 3,584 | 0.560 | 0.89 |
| hot64k8 | 47.7 ms | 2,048 | 0.344 | 0.55 |
| hot32k4a (added) | 21.7 ms | 4,096 | 0.631 | 1.05 (v2 in the same session 0.602) |
| hot64k4a (added) | 43.3 ms | 4,096 | 0.609 | 1.01 |
| hot96k4a (added) | 64.9 ms | 4,096 | 0.614 | 1.02 |
Reading: the verifier gets cheaper with `k`, because a hot load is one table read where a dataset load is an item derivation (9 mixers and 8 cache reads); the per-epoch cost is the fill, 24 to 73 ms on one core, against a 3,600 s epoch and the 20-minute seed lead. The 10 ms gate (spec 1.16) is unaffected.
### 6.4 The chip model with the measured g (M5 Max; the PC rows will replace it)
Chip A (no SRAM for `H`): gain after = `G0 / g`. Chip B (`H` in SRAM): `G0 x A_die / (A_die + A_S)` at 0.39 mm^2 per MiB, 100 mm^2 die.
| Class | g (M5 Max) | Chip A, gain after as a share of G0 | Chip B, share of G0 at 100 mm^2 | Chip B at 750 mm^2 |
|---|---|---|---|---|
| hot32k4 | 1.22 | 0.82 | 0.89 | 0.98 |
| hot64k4 | 1.12 | 0.89 | 0.80 | 0.97 |
| hot96k4 | 1.05 | 0.95 | 0.73 | 0.95 |
| hot64k8 | 1.71 | 0.58 | 0.80 | 0.97 |
The added form (second Mac session, 21:03 to 21:19 UTC, load average 7 to 14; v2 in that session 27.63 Metal, 27.59 OpenCL): the GPU keeps 0.93 / 0.87 / 0.83 of its rate at 32 / 64 / 96 MiB (`k = 4`), against the probe's 0.96 / 0.94 / 0.93, so the hot hits cost this card more than the probe says as the table grows, and 96 MiB is not resident. The named chip's position under the added form, same assumptions:
| Chip | Hot loads served from | Rate against its own version 2 rate | Gain after, as a share of G0 (S = 32 / 64 / 96) | Arithmetic |
|---|---|---|---|---|
| A, no SRAM for H | DRAM | at most 16 / 20 = 0.80 (20 dependent DRAM loads per iteration in place of 16) | 0.86 / 0.92 / 0.96 | `0.80 / g` |
| B, S MiB of SRAM for H, 100 mm^2 die, 0.39 mm^2/MiB | SRAM (the k reads near free) | 1.00 | 0.96 / 0.92 / 0.88 | `(1 / g) x A_die / (A_die + A_S)` |
| B on a 750 mm^2 die | SRAM | 1.00 | 1.06 / 1.12 / 1.15 | the SRAM is 1.6 to 4.8% of the die, the GPU's loss is larger |
With the PCs' `g` (added form, `k = 4`): RTX 5090 0.87 / 0.85 / 0.84, RX 9070 XT 0.84 / 0.81 / 0.80 at 32 / 64 / 96 MiB.
| Chip, added form, S = 32 / 64 / 96 MiB | Gain after as a share of G0, 5090 g | 9070 XT g |
|---|---|---|
| A, H from DRAM (rate 0.80 of its own) | 0.92 / 0.94 / 0.95 | 0.95 / 0.99 / 1.00 |
| B, S MiB of SRAM, 100 mm^2 die, 0.39 mm^2/MiB | 1.03 / 0.94 / 0.87 | 1.06 / 0.99 / 0.91 |
| B, 750 mm^2 die | 1.13 / 1.14 / 1.17 | 1.17 / 1.20 / 1.22 |
Reading: on the cards the plan named, the added form taxes no chip. A chip that serves H from its DRAM loses at most 8% of its gain on the 5090's figures and nothing on the 9070 XT's; a chip with the SRAM comes out ahead on every row except the smallest die at 64 and 96 MiB. The replaced form helps the recompute chip outright (section 2.3). The hot hits are not free on a GPU while the 1 GiB dataset streams through the same cache, and the layer's premise (section 1) needed them to be.
**Recommendation (owner: the project lead, gate 1):** do not adopt layer 5 in either form on these measurements. What could change it: a GPU-side way to keep H resident (cache partitioning or persisting-access controls exist on NVIDIA, approximate, from memory, and are a driver setting, not a consensus rule), or a dataset access pattern that bypasses the cache; both are outside the hash and were not measured. The experiment stays behind the flag with its packs and vectors as the record.
Reading, replaced form: on the M5 Max figures the chip's cheaper way out of `hot64k8` is the SRAM (0.80 of `G0` at a 100 mm^2 die) rather than serving `H` from DRAM (0.58). The layer's value is then the SRAM area, which is small on a large die (0.97 at 750 mm^2). The strongest configuration on this card is the one whose table fits the cache and whose `k` is large; whether 64 MiB fits the 5090's L2 and the 9070 XT's Infinity Cache with the dataset streaming beside it is the PC measurement.
## 7. Decision and what would make it pay (the Counter ASIC 3.0 note)
Decided by the coordinator on 5 October 2026 after the PC rows: layer 5 is out of v3. The rule was a GPU cost under 3% (`g` above 0.97) for a chip cost worth having; measured `g` 0.84 to 0.87 on the RTX 5090 and 0.80 to 0.84 on the RX 9070 XT for the added form, and the replaced form helps the recompute chip. No re-run.
What would make a hot table pay, for a later round:
| Condition | What the measurements say | What would have to be shown |
|---|---|---|
| A table that stays resident beside a streaming 1 GiB | The probe's ceiling at every S inside the cache (5090: 112.6 G loads/s at 32, 64 and 96 MiB) and the hash's 1.02 to 1.08x say the dataset's random lines evict it; 32 MiB on the M5 Max kept 92% of its probe gain, the 5090 kept 29% | A size small enough to survive: the probe cannot say which (it has no competing stream); a sweep of `S` down from 32 MiB (16, 8, 4, 2) with the hash itself, on the 5090 first, would. A 2 to 4 MiB table costs a chip under 2 mm^2 of SRAM, so the layer would then tax nothing; the point of the layer was a table a chip cannot afford, and a table a GPU keeps is one a chip affords |
| The `k` and `S` the probe says | `g` moved with `k` (1.20x at `k = 8`, 1.02 to 1.08 at `k = 4`, 1.00 at `k = 2`) and barely with `S`; the probe predicted 1.27 to 1.73 | Only a resident table makes `k` worth raising; at `k = 8` with a resident table the replaced form reaches 2x (ideal) and the added form costs 0 by the probe. Without residency, `k` buys rate on the replaced form (which helps the chip) and costs rate on the added form |
| A different access shape | Both forms read H at one random word per hot load, the dataset's own pattern, and share the cache with 128 dataset reads per hash | A shape that touches the table in cache-line units and few lines per hash (one 64-byte line per iteration, say) would need a smaller resident set per hash and could be measured with the read-width emitters (`wide_load_stmt`) pointed at H. Or a hot table whose lines are read in a fixed order per epoch (a stream, not a random read), which a GPU prefetches and a chip must still hold or fetch; not designed here |
| Cache partitioning on the GPU | NVIDIA exposes persisting L2 access controls to CUDA programs (approximate, from memory); AMD and Apple do not expose an equivalent to OpenCL or Metal | A consensus rule cannot depend on a driver feature of one vendor; it could be a miner-side optimisation if the layer were in, which it is not |
What the round leaves in place: the code behind the flag (`LoadClass::hot`, both forms, the hosts filling H on the device from the epoch seed, the eight packs and their vectors), the probe at the hot sizes on three cards, and the measured rule that a read-only table a GPU cache could hold is not one it does hold while the dataset streams.
## 8. What is unverified
Listed here until measured, and carried into the bench-log entry.
1. Measured on all three cards (6.2): no card keeps `H` resident enough to deliver the probe's hit rate while the dataset streams. Not measured: whether a driver-side cache partition (NVIDIA's persisting L2 access controls, approximate, from memory) would; it is not something a consensus rule can rely on.
2. The PC rows are one run each (jobs run-ca2-hot-5090-20261005 and run-ca2-hot-9070-20261005, 5 October 2026); the two launch shapes per card agree within 1%, and no re-run was taken (PC 1 was handed on).
3. The resident warp counts of section 3 are approximate; the occupancy the hosts reach is printed by each harness (`kernel:` lines) and should replace them.
4. The SRAM area figures are approximate (section 4 cites their sources); no chip was priced.
5. `S = 96` uses the multiply-shift mapping; the spec's dataset still uses `AND MASK`. If the layer goes in, gate 1 decides whether the dataset mapping follows (section 1.13.3) or `S` stays a power of two.
6. The hot table on the chain needs the epoch seed about 1 ms (GPU) or up to 71 ms (CPU) before the epoch starts; the 20-minute lead of section 4.3 covers it. Not exercised on a node.
7. The Mac numbers were taken at load average 14 to 27 (other agents' CPU work); the ratios between packs of one session are the result, the absolute Mhash/s are not quiet numbers.

171
docs/plans/miner-ui-2.md Normal file
View file

@ -0,0 +1,171 @@
# Igneum Miner UI 2: sections behind a rail, 5 October 2026
the project lead, 18:40 UTC: "can any updates be made to the UI of the miner? different sections for different features. Made
super simple for users, all settings super easy and simple and everything looking gorgeous, as close to shippable
as we can get right now." Worktree `/Users/joshm/Projects/igneum-wt-miner-ui`, branch `miner-ui-2` from master
a93199a. the project lead approved the redesign from the screenshots (20:1x UTC) and moved it into 0.3.10: the 0.3.10 shipper merges
`miner-ui-2` last and rebuilds the app. Times are UTC.
## 1. What changed
| Where | What |
|---|---|
| `app/igneum-app/ui/index.html` | One page became a rail with six sections: Mine, Prove, Rewards, Node, Updates, Settings. The welcome, GPU and address screens and the one-time key sheet stay as they were. The settings panel, the bottom bar and the proving tile are gone; every control they held is placed on a section. |
| `app/igneum-app/ui/app.css` | Rewritten around the rail, the thin top bar, the status strip and the pages. Same tokens as the site (obsidian, graphite, ember, molten, bone, ash; Unbounded, IBM Plex Sans, IBM Plex Mono). One accent. Lays out from 900 x 600: under 1000 px the rail folds to icons, the number strips to two columns, the two-column grid to one. Focus rings on every control (`:focus-visible`, the switch tracks). |
| `app/igneum-app/ui/app.js` | New pure block `View` (the words every section shows for a state), tested by `view.test.mjs`. The Notices and UpdateCard blocks, the log drawer and the blocks canvas are unchanged. `?page=<name>` forces a section and `?logs=1` opens the drawer (screenshots). |
| `app/igneum-app/ui/view.test.mjs` | 10 tests: the GPU row (names, kinds, integrated off, temperatures amber then red, unknown numbers blank), the big button, the node words, the next switch, the prove words, the dev-fee lines, the jobs line. Added to `.github/workflows/ci.yml`. |
| `app/igneum-app/src/state.rs`, `engine.rs` | Engine addition 1: `node.consensus_digest`, the node's own "Consensus params digest" line (`digest_from_line`, tested; a peer's digest in a WARN line is never taken). Addition 2: `node.consensus_switches`, every `*_activation_daa` in the override file the node was started with, lowest first, in plain words (`switches_of`, tested). |
| `app/igneum-app/src/prover.rs` | Engine addition 3: `proving.program_id` and `proving.aggregator_id` from `igneum-prove-host --mode id` (`ids_from_describe`, tested), read once at thread start on macOS and Linux and when the WSL2 host is found on Windows, proving on or off. |
| `app/mac/IgneumMiner.swift` | Window minimum 900 x 600 (was 900 x 620). |
| `app/windows/host.cpp` | One `WM_GETMINMAXINFO` case: minimum 900 x 600 (there was no minimum). |
The API contract is otherwise as it was: the UI calls the same routes (`api/state`, `api/cards`, `api/pause`,
`api/resume`, `api/prove`, `api/settings`, `api/update/*`, `api/jobs/*`, `api/sweep/*`, `api/key/*`, `api/log`,
`api/open`, `api/quit`, `api/clock/sync`, `api/power/apply`, `api/detect`, `api/phase`, `api/setup`, `api/start`).
## 2. The sections
| Section | What it holds | Where it was |
|---|---|---|
| Mine | One big Start mining / Stop mining button (pause and resume), hash rate, blocks found, next program; one row per GPU with its real name, kind tag (Integrated shown as such and off by default, the engine's rule), on/off switch (applies at once; the other cards keep their identities and caps), hash rate, temperature and power where the card reports them; the blocks strip; Activity | the dashboard strip, the Cards card, the Pause button, the Events card |
| Prove | The proving switch with three plain sentences on what proving is; state, assigned, proven, paid (IGN when paid); the verifier's state with one line; the shard program id and the aggregator id with Copy; Set up when the WSL2 host is missing | the Proving tile and the Settings switch |
| Rewards | The address with one Copy; blocks found, this run, balance; the Save your key card (two lines, Show my key, Hide) or the "you pasted your own address" line; Use another address; the dev-fee line; the wallet page | the Settings address block |
| Node | Node state, height, peers, version, each with a plain line; the chain numbers; the consensus digest with Copy; the next switch with its height and how far away; every switch with the applied ones dimmed; finality; the clock card when the clock is off | the Node and Finality cards |
| Updates | The version, Check now, Install now when a download is ready, the auto-update switch; remote jobs: Check now, the state line, the history table, the signing key | the Settings version and remote-jobs blocks |
| Settings | Per card: power cap slider with the watt number (NVIDIA; Apple silicon says it manages its own power), identities stepper, sweep line with Sweep now / Stop / Unpin, the cap-applied or Retry line, draw and temperatures; the sweep switch; start at login; remote jobs allowed; vote on checkpoints; the machine name; the dev fee stated plainly; Copy the log and Show the log, the log and chain folders; Advanced (collapsed): the devnet trust switch, the live page | the Settings panel |
| Status strip | Unchanged (Notices): updates, remote jobs, the clock, one at a time, under the top bar | the same place |
| Rail foot | Machine name, rewards address, Logs (the drawer), Quit, version and chain | the bottom bar |
Every control in Settings is one switch, one slider or one field with one line of help under it. Nothing was
removed: the identities stepper, the sweep buttons, the power Retry, the trust switch, the live page, the machine
rename, the key reveal, the log drawer with its chips, search, ruler and jump controls all still exist.
Empty, loading and error states: Mine says "Asking the graphics cards to report in" then "No GPU this app can drive
was found" (red) with the detect message; the big button is disabled with a reason while no card is on, the engine
is away or the app quits; Prove says off, needs setup, waiting, proving, submitted or idle, each with a sentence;
Node says starting, syncing with a percentage and the ETA, synced, restarting, failed or stopped, and the digest
box says "not printed yet" or "the node is not running"; Updates says "Not checked yet" or "This build has no update
address"; the jobs table says "No job has run on this machine yet"; Settings says "No card yet"; a lost engine
turns the pill to "engine away" and the Mine and Node words to "no answer".
## 3. Proof
The engine was built from this worktree on the Mac under the build lock (`with-lock.sh build nice -n 19 cargo build
--release -j 4`, rustup's cargo 1.99; Homebrew's cargo 1.69 in PATH cannot read the lock file) and run on its own
port and data directory in devnet v4 mode (`IGNEUM_APP_BIN` = the installed 0.3.9 bundle's `Resources/bin`,
`IGNEUM_APP_DATA`, `IGNEUM_APP_LOGS`, `IGNEUM_APP_NODE_DIR` under the session scratchpad, RPC 27610, P2P 27611,
a scratch `igneum-app.json` next to the binary with the devnet override params, no update manifest) through
`with-lock.sh run`. The live Mac app's engine, node 1 and the observer were not touched; the test node peered with
the seed and node 1 and synced 123,559 blocks in about 12 minutes, then the M5 Max mined on it (22 MH/s, digest
1f4b4425..., the devnet's). The setup flow was walked in the built-in browser pane (Get started, the GPU switch,
Make me an address, the key sheet, Start mining) and every section was clicked through, the rail driven with the
arrow keys, and the window checked at 900 x 600.
Tests: `node --test notices.test.mjs update-card.test.mjs view.test.mjs` = 20 pass, 0 fail; `cargo test --release
--bin igneum-app -- digest_comes switches_are pinned_ids` = 3 pass (the three new engine tests, run on the Mac
under the build lock; the full suites go to PC 2 with the 0.3.11 cut as usual).
The PNGs below were taken with the installed Mac host's snapshot mode (`"Igneum Miner" --snapshot <png> --url <the
test engine> --size 1200x780`, a real WKWebView, Retina 2400 x 1560), in `docs/plans/miner-ui-2/`.
| File | Shows | Placeholder or sample |
|---|---|---|
| `00-welcome.png` | The welcome screen (unchanged) | none |
| `01-setup-gpu.png` | Step 1, the M5 Max row with its switch | none |
| `02-setup-address.png` | Step 2, make an address or paste one | none |
| `03-setup-key.png` | The one-time key sheet | the key is zeros (`?screen=key` sample) |
| `04-mine.png` | Mine while mining: the big button, 23 MH/s, the GPU row, the blocks strip | temperature and power say n/a on Apple silicon (the engine has no reading for it) |
| `05-prove-off.png` | Prove with proving off, the verifier verifying, both pinned ids | none |
| `06-prove-on.png` | Prove with proving on: Proving (CPU), block 35507 shard 0 | none (the Mac CPU prover was switched on for 25 s, then off) |
| `07-rewards.png` | Rewards: the address, 343 lifetime blocks, Save your key, Use another address | the balance cell says "--, shown in the wallet, not here yet" (no API) |
| `08-node.png` | Node synced: height, 2 peers, version 2.1.0, the digest, next switch Fees v1 at 210,000, finality | none |
| `09-updates.png` | Updates: 0.3.9, Check now, auto-update, remote jobs | this build has no manifest, so "no update address" and "no jobs address" |
| `10-settings.png` | Settings: the card block, the sweep switch, this machine, dev fee, logs, Advanced collapsed | the NVIDIA power slider is not in the shot (no NVIDIA card on this Mac); its markup is in `setCardHtml` |
| `11-logs-drawer.png` | Mine with the log drawer open (chips, search, ruler, follow, close) | none |
| `12-strip-update.png` | The status strip with "Igneum Miner 0.3.7 is available", Install now, Later | `?update=available` sample |
| `13-strip-job.png` | Updates with a running remote job in the strip and the table | `?job=running` sample |
| `14-node-syncing.png` | Node while syncing: 59,602 of 123,559 (48%), the next switch from that height | none |
| `15-mine-syncing.png` | Mine while the node syncs: the button says Stop mining, waiting for the node to sync | taken before the blank marker in the GPU row became `n/a` (it shows a dash) |
| `16-narrow-900-mine.png` | Mine at 900 x 620 (the installed host's minimum; the new hosts say 600): the icon rail, two-column numbers, the power column dropped | none |
| `17-narrow-900-settings.png` | Settings at 900 x 620 | none |
| `18-narrow-900-node.png` | Node at 900 x 620 | none |
## 4. Shippable now, placeholder, follow-ups
Shippable now: the six sections, the rail, the strip, the drawer, every setting, the three engine fields, the
hosts' minimum size. The Windows host line is untested on Windows (one `WM_GETMINMAXINFO` case; the 0.3.11 cut
builds it on the runner as usual). The NVIDIA power slider, telemetry and sweep lines on Settings and the
temperature and power columns on Mine are rendered from the same state fields the old tiles used, but no NVIDIA
card was on this Mac, so they are unseen in these shots: the PC screenshot is the first thing to look at after
the cut.
Placeholder: the balance cell on Rewards ("shown in the wallet, not here yet").
Follow-ups (engine work, not built here; the budget was three small additions):
| Follow-up | What it needs |
|---|---|
| Balance on Rewards | the engine reads `eth_getBalance` for the rewards address from the node's EVM RPC every 30 s (`update.rs` has the curl shape) and puts `address.balance_wei` on the state |
| "Pause while I use the machine" | does not exist in the engine: an idle-input reading per platform and a pause/resume on it; the Settings switch is one line once the field exists |
| Log export as a file | today Copy the log puts the last 2,000 ring lines on the clipboard and the folder path is shown; a `/api/log/export` that writes the ring to the log folder and reveals it in Finder or Explorer needs a `platform::reveal_path` |
| Temperature and power on Apple silicon | the engine reads no telemetry for Metal; `powermetrics` needs root, so this stays blank unless a non-root source is found |
| A per-card "pause this card" without restarting the others | exists already through the row switch (only that card's worker restarts); nothing to do unless the project lead wants a timer |
## 5. The 0.3.10 merge: gpu-hotplug (4d122e1) resolved on the new UI
The branch was merged with `gpu-hotplug` 4d122e1 (the 0.3.10 tree carries it). The engine, relay and tooling files
merged by themselves; `app.js` and `app.css` were taken from this branch and the hot-plug UI ported by hand:
| Hot-plug state | On the six-section UI |
|---|---|
| The strip notices (card added, mining; new card not usable (Code 43); card removed) | the Notices block is gpu-hotplug's, unchanged; `notices.test.mjs` (12) passes |
| A card the OS reports a problem on (`problem`, "Code 43") | Mine row: name in ember, the word "not usable (Code 43)", the reboot hint under it, the switch disabled; the Mine note repeats the hint; first-run row and Settings card say the same, with no controls |
| A removed card (`removed_at`) | Mine row dimmed, the word "removed", "unplugged; its worker stopped. The row goes in five minutes."; no switch; Settings card says the same; after five minutes (`gone`) the row is not shown |
| The counts and the big button | only present cards count (a removed or faulty card never makes the button say "Stop mining") |
| The name tooltip | the tool's code (gfx1201), the device index, the OpenCL platform, the PCI address (`View.cardTitle`) |
| The card list sent on a switch | removed and faulty cards are left out, as gpu-hotplug's `readCardRows` does |
`view.test.mjs` has an eleventh test for the two states. The whole app crate's unit tests pass on the merged tree
(99, on the Mac under the build lock, `cargo test --release --bin igneum-app`). `relay/test/parse.test.mjs`
passes. The Mac engine builds and runs (the same scratch instance as section 3).
Build tooling for the Windows compile: `packaging/windows/push-build-inputs.sh --no-node` and
`node tools/build-job.mjs run --no-node` pack and build the app engine only (no node source, no node build, no node
tests), so an app-only change compiles on PC 1 in a fraction of the full job.
PC 1 job (20:21Z, from this tree at d62b445, `node tools/build-job.mjs run --target ae432dc7 --no-node --targets
windows --no-tests --no-place`): extract 167 files, `windows app/igneum-app build exit 0 6 s` (the PC's target dir
was warm), `igneum-app.exe` 2,970,112 bytes sha256 62ce96a9..., PE check ok, coin icon and version block ok, done
after 13 s. The exe carries the new UI and engine strings ("Prove shards on this machine" x2, `consensus_switches`
x6, "Consensus params digest:" x1, `card:removed:` x1), so the 6 s was a real compile of the merged sources, with
gpu-hotplug's Windows-only `detect::adapters` path and this branch's prover change in it. Downloads under the session
scratchpad (`pc1-app/`), not placed.
The ship tool runs no node tests of its own (`tools/ship-app.mjs --self-test` is the version bump), so the only
list to carry `view.test.mjs` is `.github/workflows/ci.yml` (done in 83a293f).
What the PC job does not compile: `app/windows/host.cpp` (the window host with the `WM_GETMINMAXINFO` and
`WM_DEVICECHANGE` cases) is built by the GitHub runner (`windows.yml`) at the shipper's push, never on a PC; the
`WM_GETMINMAXINFO` case is three lines of plain Win32 (`MINMAXINFO`, `ptMinTrackSize`) and reads at the runner's log.
## 6. The changelog paragraph for the 0.3.10 manifest notes (what the user sees)
The miner has six sections behind a rail: Mine, Prove, Rewards, Node, Updates, Settings. Mine has one big Start
mining / Stop mining button, the hash rate, the blocks found, and one row per graphics card with its name, an on/off
switch, its hash rate, temperature and power. Prove says in plain words what proving is and shows the shards
assigned, proven and paid, the verifier and the program ids. Rewards shows the address with one Copy and keeps the
key backup in its own card. Node shows the sync, height, peers and version with a plain line under each, the
consensus digest, and the list of planned rule switches with the next one and its height. Updates holds the
version, Check now and the remote jobs. Settings has one switch or slider per setting with one line of help under
each: the power cap per card with the watt number, identities, start at login, remote jobs, voting, the dev fee
stated plainly, the log. The window lays out from 900 x 600. Not in this release: the balance on Rewards (shown in
the wallet), a "pause while I use the machine" switch, log export as a file, temperature and power on Apple
silicon.
## 7. The branch
| Commit | What |
|---|---|
| 83a293f | the UI pass, the three engine additions with tests, the hosts' minimum size, this plan and the 19 screenshots (the blank marker in a GPU row is `n/a`, not a dash: copy law) |
| 364feef | this plan's hash line |
| d62b445 | Merge gpu-hotplug 4d122e1: the hot-plug states on the new UI, `--no-node` for the build tools |

Binary file not shown.

After

Width:  |  Height:  |  Size: 316 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 153 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 196 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 242 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 281 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 338 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 327 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 302 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 340 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 245 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 312 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 455 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 284 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 274 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 342 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 277 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 76 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 231 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 166 KiB

418
docs/plans/mixer-x4.md Normal file
View file

@ -0,0 +1,418 @@
# Mixer x4 and the cache growth rule: the class v3 dataset construction
5 October 2026 (night). Counter ASIC 2.0, layer 6 (option C) and ledger M16's lever, decided by the coordinator
under the project lead's delegation at 22:00 UTC (`docs/plans/counter-asic-2-status.md`, "22:00 decided"; the project lead confirms for
the public testnet genesis). Branch `ca2-mixer`. Worker: ca2-mixer (cryptographer's lane).
What this changes, in one line: under program class v3 every mixer application of the dataset item derivation
becomes four applications with distinct round keys, the eight dependent cache reads per item stay eight, and the
256 MiB cache doubles on the days the dataset doubles (years 4 and 12). Version 2 is byte-identical: the two
pinned packs re-export without a changed byte (section 5).
## 1. Why this form
The recompute attacker of `docs/analysis/m16-recompute-attacker-2026-10-05.md` holds the 256 MiB cache on a die
and derives every dataset word instead of reading it: 128 items per hash at about 1,170 integer operations and 8
dependent cache reads each. Its cost is linear in operations per item; the honest miner pays the mixer once a day
in the dataset build and never per hash; the verifier pays it per item it checks. The multiplier `m` is the one
parameter that moves the attacker and leaves the honest hash rate untouched.
Two shapes give the attacker 4x the operations:
| Shape | Mixer applications per item | Dependent cache reads per item | What grows for the verifier | What grows for the chip | What grows for the honest build |
|---|---|---|---|---|---|
| A, chosen: `m = 4` applications per round, 8 rounds | 36 | 8 | the ALU part only; the latency part (8 dependent misses per item) unchanged | integer operations 4x; SRAM bandwidth unchanged (1,024 reads per hash) | 4x the mixer arithmetic, same reads |
| B, alternative: 32 rounds of one application and one read | 33 | 32 | both parts: 4x the dependent misses per item, so about 4x the latency-bound time (spec 1.11: about 8 x 100 ns per item in series without interleaving) | integer operations 3.7x and SRAM bandwidth 4x (4,096 reads per hash, 60 TB/s to match one 5090 at the M16 rate) | 4x the reads too; the GPU build becomes latency-bound at 4x the dependent line fetches |
Shape A is chosen because the verifier's latency part is the part the 10 ms gate protects (section 1.11: the
distinct items of a unit are derived with their chains interleaved so the 8 misses of each item overlap across up
to 32 items; shape B would make that 32 misses deep). Shape B is implemented nowhere; the rule in the brief: if
shape A's measured verifier time exceeds 4.8 ms per warp on one M5 Max core, measure both and recommend. Section 6
has the measurement; it is under that bound, so B stays unimplemented.
## 2. Spec text (replaces 1.8.5 and 1.13.3 under class v3; v2 text unchanged)
### 1.8.5 Item derivation and dataset mapping
Item `t` (16 words) under mixer multiplier `m` (`m = 1` for program class v2, `m = 4` for class v3; a class
parameter, `LoadClass::mixer_mult`):
```
s[i] = K[i] for i in 0..7
s[8 + i] = t * MUL[i] + RC[i] for i in 0..7
for r in 0..7:
for j in 0..m-1:
s = M(s, rk = (r * m + j + 1) * 0x9E3779B9)
a = s[0] AND (2^(C - 4) - 1) cache line index, 2^(C - 4) lines of a 2^C-word cache
s[i] = s[i] XOR cache[line a][i] for i in 0..15
for j in 0..m-1:
s = M(s, rk = (8 * m + j + 1) * 0x9E3779B9)
item(t) = s
```
`M(s, rk)` is the mixer of 1.8.4 with round key `rk`; under `m = 1` the keys are `(r + 1) * 0x9E3779B9` and
`9 * 0x9E3779B9`, the version 2 text exactly. The round keys of the `9 m` applications are the first `9 m` values
of the version 2 key sequence, all distinct (the sequence is `k * 0x9E3779B9` for `k = 1 .. 9 m`, and
`0x9E3779B9` is odd, so no two of the first 2^32 keys coincide). Eight dependent cache reads per item at every
`m` (`ITEM_ROUNDS = 8`, prototype value): the address of read `r` depends on every earlier read. `9 m` mixer
applications, 36 under class v3, about 4,700 integer operations per item (130 per application, 1.8.4).
`dataset[w] = item(w >> 4)[w AND 15]`. A dataset of 2^D words is the prefix of items `0 .. 2^(D-4) - 1`, so an
item has the same value at every dataset size; and a cache of 2^C words is the prefix of segments of every
larger cache (1.8.3 fills segments independently of the cache size), but an item's value depends on `C` through
the line mask, so the item changes on the day the cache doubles.
Source: `igneum-pow/src/memhard.rs` (`derive_items`, `round_key_mult`, `Shape`), the emitted `mh_item` of
`memhard.h`, `memhard.metal` and `kernel.cl` (`emit.rs`, `emit_memhard_core`: the `m` loop is emitted only for
`m > 1`, so every version 2 pack keeps its text).
### 1.13.3 Dataset growth (class v3: option (b) with the cache tied to it, "option C")
Designed: 2 GiB at genesis plus 0.5 GiB per year. The linear schedule in bytes, `G x (1 + 86,400 d /
(4 x 31,536,000)) = G x (1 + d / 1,460)` for the genesis size `G` and the chain day `d` (DAA days since genesis,
section 1.12), doubles at day 1,460 (year 4), quadruples at day 4,380 (year 12), reaches 8x at day 10,220
(year 28). Rule (Designed, decided 5 October 2026 for class v3):
```
doublings(d) = floor(log2(1 + d / 1460)) integer division, then integer log2
dataset_words(d) = 2^(D_0 + doublings(d)) D_0 = 29 designed (2 GiB), 28 on the devnet (1 GiB); capped at 32
cache_words(d) = 2^(26 + doublings(d)) 256 MiB, 512 MiB from year 4, 1 GiB from year 12
```
Power-of-two sizes only (option (b)), so every load keeps the `src AND MASK` form of 1.14 item 2 and the cache
line index keeps `s[0] AND mask`. The cache doubles exactly when the dataset doubles ("option C",
`docs/analysis/sram-mirror.md` section 7): the cache's job is to stay above any GPU's last-level cache and that
needs growth; the recompute attacker is priced by the mixer, not by the cache (section 7 below).
`d` is `day_index(header.timestamp) - day_index(genesis.timestamp)` with `day_index = timestamp_ms / 86,400,000`
(`bind::day_index`, the interim day rule), clamped at 0 (`memhard::days_since_genesis`). Under class v2 nothing
grows: the cache is 2^26 words and the dataset the genesis size on every day.
| Chain day `d` | Years | `doublings` | Cache words | Cache | Dataset words (devnet `D_0 = 28`) | Dataset (designed `D_0 = 29`) | Verifier cache fill, one M5 Max core (measured at 256 MiB, section 6, scaled linearly) |
|---|---|---|---|---|---|---|---|
| 0 to 1,459 | 0 to 4 | 0 | 2^26 | 256 MiB | 2^28 (1 GiB) | 2 GiB | 0.18 s |
| 1,460 to 4,379 | 4 to 12 | 1 | 2^27 | 512 MiB | 2^29 (2 GiB) | 4 GiB | 0.36 s |
| 4,380 to 10,219 | 12 to 28 | 2 | 2^28 | 1 GiB | 2^30 (4 GiB) | 8 GiB | 0.72 s |
| 10,220 to 21,899 | 28 to 60 | 3 | 2^29 | 2 GiB | 2^31 (8 GiB) | 16 GiB | 1.4 s |
| 21,900 and on | 60 and on | 4 | 2^30 | 4 GiB | 2^32 (16 GiB, the index cap) | 2^32 words, the cap | 2.9 s |
Test: `memhard::tests::growth_schedule_table` pins every row and the day before each step. The devnet pack
`igneum-devnet-v4-epoch0` is day 20,730 of the Unix count against genesis day 20,729, `d = 1`, so every existing
size and vector stands.
Consequences for the tiers (the rule of 5 October): a verifier (any node, any pool core) holds 512 MiB from year 4
and 1 GiB from year 12, and fills it once a day in under a second on one 2026 core (the table); a miner's card
holds the dataset, 4 GiB from year 4 and 8 GiB from year 12 on the designed schedule, so an 8 GB card mines until
year 12 and a 16 GB card until year 28 (the cache is not in the card's working set at hash time: it is built,
the dataset built from it, and dropped). Those dates are the design document's own schedule restated as steps;
option (a) would have faded a 4 GiB card out in year 4 instead of year 4.
## 3. Interfaces
| Item | Where | Note |
|---|---|---|
| `LoadClass { mixer_mult: u8, growth: bool }`, `LoadClass::MX4` ("mx4"), `with_mixer(m, growth)`, `v2_loads()`, `takes_width_roll()` | `generator.rs` | a class with v2 loads takes no width roll: its program stream is version 2's draw for draw, so the v3 program of a seed is the v2 program of that seed, only the dataset differs |
| `Shape { mixer_mult, cache_log2_words }`, `Shape::for_class_day(class, d)`, `MixParams.shape`, `Cache::fill_log2(key, log2)`, `round_key_mult(r, j, m)` | `memhard.rs` | `Shape::V2` is version 2 |
| `growth_doublings(d)`, `cache_log2_words(d)`, `dataset_log2_words(D_0, d)`, `days_since_genesis(day, genesis_day)` | `memhard.rs` | the schedule, one function and its two sizes |
| `DatasetSource::{new_shape, from_key_shape, shape}`, `Epoch::new_class_day`, `Epoch::from_seed_bytes_day(epoch, day, label, class, d, D_0)` | `verify.rs` | the day-sized entries; the v2 entries are unchanged and build the v2 shape |
| `IGNEUM_MIXER_MULT`, `IGNEUM_CLASS_MIXER_MULT`, `IGNEUM_CACHE_GROWTH` in program.h; `"mixer_mult"`, `"cache_growth"`, the `"item"` string in program.json | `emit.rs` | written only for a class with `m != 1` or growth, so v2 packs do not change |
| `packfile.h` `mixerMult`; `packbench` and the OpenCL host print the multiplier and the cache size | the three hosts | the kernels carry the construction in their text (one emitter, three dialects); the hosts size the cache from `IGNEUM_CACHE_LOG2_WORDS` already (packbench.swift line 56, host.cu line 65, host.c line 1966) |
| `igneum-pow --class mx4 [--days d]` on every command | `main.rs` | `--days` sizes the cache for a growth class |
Under the ca2-v3 seam (`ProgramClass::V3`, `V3_CLASS`), the integration sets `V3_CLASS = LoadClass::MX4`; the
chain's day-sized dataset needs the day index, so `Epoch::chain_dataset(day, class)` builds the genesis-size
cache and a `chain_dataset_day(day_bytes, class, d, D_0)` beside it is the growth entry (section 9, owed to the
node agent).
## 4. Vectors (class v3, `proto-cuda/packs-ca2-mixer/`)
Produced by `igneum-pow export --program-class v3` (the Rust CPU interpreter, 5 October 2026, commit 66eeba3) and
checked on the GPUs in section 6. The program of each pack is the version 2 program of the same seed instruction
for instruction (`tests/packs.rs`, `v3_packs_are_the_v2_seeds_under_mixer_x4`); the cache is the version 2 cache
(day 0 of the growth rule); the dataset words and the hashes are new. The 64 sampled indices are those of every
pack (`emit::sample_indices`).
Pack `mx4-genesis` (seed igneum-genesis, day 2026-10-03, generator 3, class mx4, program id e323b9dcaf283a6f, 2^28 words, 2^26-word cache, cache FNV-1a 64 `48c4f5bf24166b2e` as under v2):
```
dataset words 0..15 (item 0):
61ff2180 0d4c7e6c 2177d443 60df9025 cf8b2e10 63675bfb 25289e58 9c45dc42
2d271c54 9652369b 2dd77508 5921392c 3afa60ee c640ad68 f2bb56ff cfa46438
dataset[0x0fffffff] = 5020180e
sampled words (the first 8 of the 64 in vectors.json):
dataset[59471966] = de85726d dataset[217795994] = 7cfc31c7 dataset[208353206] = d3cd5289 dataset[42483309] = 6ebbeef1
dataset[172547758] = 9858413b dataset[148076330] = e786f141 dataset[183853158] = 64f13833 dataset[214389424] = ca229d04
unit at base nonce 0, lanes 0..31:
63acd2d273f475ba e929c78b34b80d4b 0b1011cb19982558 1457a0df5497aa11
957d0f3bb71d98fb ac16901e6e6f6057 8ea1c6279f4b177a f28146e60bd08ba9
fe5b8cfe87f8e65b 49f87240566ace62 6ef6d6b7bdea8e41 46d9c0dc29a97b9c
111fe30128db9398 66dc39084f0946d4 8ee11bdfd35fecf2 2861fcfc75db6677
31c7667d4bde8556 c5989c48858b4ce0 276395e734a9d30d 84217b41e91368ff
3604861e34d9f697 9f51d8ee16bf3639 c89e47bafa84401c 7ae78c1f10b70e19
0b8c947157a29a48 d67192e8cfb43842 05a4c6d182c8c675 188e2661f3263f2e
a2df24238f7fea2e ed69ea7e13ad3a48 2d44ae509bab91b8 adad61931ea4fb70
unit at base nonce 4096: lane 0 edd508ac57e5699a, lane 1 4eefd56d526cdaeb, lane 31 8892f8604733b1e0
unit at base nonce 1000000: lane 0 8b3183778a49f59c, lane 1 1831b72a8797e895, lane 31 75eae55eba53a506
```
Pack `mx4-devnet-epoch0` (seed the devnet genesis hash, day bytes of 2026-10-04, generator 3, class mx4, program id 73bcbfe8ccf988f1, 2^28 words, 2^26-word cache, cache FNV-1a 64 `448274a57f508cbc` as under v2):
```
dataset words 0..15 (item 0):
afe80d67 b9fbd029 6c79f193 95139ad9 96310aff 4609f8b1 75279e63 28235be1
47b17dcb 718e0ef2 a52588c8 a8bf49d5 19cf243e 5ec8905e a4851f66 af9cd9f3
dataset[0x0fffffff] = e6a99c7a
sampled words (the first 8 of the 64 in vectors.json):
dataset[59471966] = 57642b58 dataset[217795994] = c279badd dataset[208353206] = cbccbaad dataset[42483309] = 32cce392
dataset[172547758] = 71fdb4c6 dataset[148076330] = f5a268ce dataset[183853158] = 3ca1d676 dataset[214389424] = 7977b03d
unit at base nonce 0, lanes 0..31:
212c6442b51e87ae c374795c00839331 b6036a220a98f4b3 eb8b8013e637367b
db5866e9b73930fd f3f3d01f46e90333 9d913991ab8ed428 7ccb1d8fa100a800
3cf45ba44f09a0fe 91acf48ef1a63082 6ea46c69fb082f99 581f0218977a9d72
9a4623a5c62ddf2d ab6eb5e768f0feb4 07b70bdccf8aca12 d666311ae5e4311e
53114757d669f0a4 bd5d6ace87ce2ce4 fb712015e8189192 a32cec81103e134b
83f3d18c3289c124 fe29f1984b132b3d c9ffcf4e3774497a ac99c9243dc63809
d78a7e8217a32f3c ab81ad63d242fc31 0e4c30b7e00024af ce014289fff6778d
64a1292e2a8b4a91 d5b8c90e681d7e3a 06078117673030fd 51bf77b280173930
unit at base nonce 4096: lane 0 3c797978566b5950, lane 1 7c759e60185b6411, lane 31 96a903eb9a0ca390
unit at base nonce 1000000: lane 0 d5a8da0568df8ee7, lane 1 68cb69c04208285a, lane 31 f8ca84a1a5d78cf5
```
## 5. The v2 path is byte-identical
`cargo test --test packs` regenerates every file of `igneum-genesis-mh` and `igneum-devnet-v4-epoch0` from
`program.json` and compares byte for byte (`emitted_sources_match_all_packs`, `export_pack_matches_all_packs`);
section 6 also records a fresh `igneum-pow export` of both packs diffed against the checked-in directories.
## 6. Measurements
Machine: Apple M5 Max, 64 GiB, Darwin 25.6.0, 5 October 2026 (night), other agents' builds running beside every
run; a timing row says which lock it ran under (`measure` is exclusive; `run` and `build` are not timings).
### 6.1 The v2 path, byte for byte (no lock needed)
`igneum-pow export --seed igneum-genesis --day 2026-10-03` and `igneum-pow export --epoch-hex edc4fa84...fb07
--day-hex 69676e65756d2d6461792ffa50000000000000` on commit 66eeba3, `diff -r` against
`proto-cuda/packs/igneum-genesis-mh` and `igneum-devnet-v4-epoch0`: IDENTICAL, both (twelve files each). The
crate tests regenerate the same files and compare them on every run (`tests/packs.rs`, 12 of 12 pass).
### 6.2 Bit-exactness of the class v3 construction on the GPUs (`with-lock.sh run`, 22:05 UTC)
`packbench --pack <dir> --batches 1 --batch-log2 24 --group 256` (Metal, built from this branch) and
`igneum-bench-cl-igneum-genesis-mh --bench-pack --pack <dir> --batches 1 --batch-log2 24` (Apple OpenCL, built from
this branch's host.c). Vectors are the Rust interpreter's; the fingerprint is FNV-1a 64 over the 2^24 outputs at
base nonce 0.
| Pack | Harness | Cache FNV-1a 64 | Dataset head, word MASK, 64 samples | Vectors standalone / in batch | Fingerprint 2^24 | MH/s (GPU time; not a measurement, the run lock) |
|---|---|---|---|---|---|---|
| mx4-genesis | Metal | 48c4f5bf24166b2e PASS | PASS | 3/3, 3/3 | 6f48d5a2aa0dbe5f | 27.6 |
| mx4-genesis | Apple OpenCL | PASS | PASS | 96 of 96 lanes | 6f48d5a2aa0dbe5f | 27.7 (wall) |
| mx4-devnet-epoch0 | Metal | 448274a57f508cbc PASS | PASS | 3/3, 3/3 | 73caaebb28e808fe | 27.5 |
| mx4-devnet-epoch0 | Apple OpenCL | PASS | PASS | 96 of 96 lanes | 73caaebb28e808fe | 27.6 (wall) |
| mx8-genesis (the x8 candidate, 21:45 UTC) | Metal | 48c4f5bf24166b2e PASS | PASS | 3/3, 3/3 | 7c28cfb06c5c65a9 | 27.7 |
| mx8-genesis | Apple OpenCL | PASS | PASS | 96 of 96 lanes | 7c28cfb06c5c65a9 | 27.6 (wall) |
| mx8-devnet-epoch0 | Metal | 448274a57f508cbc PASS | PASS | 3/3, 3/3 | bbb183f72692f840 | 27.7 |
| mx8-devnet-epoch0 | Apple OpenCL | PASS | PASS | 96 of 96 lanes | bbb183f72692f840 | 27.7 (wall) |
Reading: the Rust interpreter, Metal and Apple OpenCL agree on the v3 dataset (head, MASK word, 64 samples), on
every vector lane and on the 2^24-output fingerprint of each pack; the hash rate is the v2 rate (27.7 MH/s on this
card tonight, readwidth table), as it must be: the hash kernel only loads, the mixer is paid in the build.
Dataset build under the run lock (indicative only; the measured rows are in 6.4): Metal 30.2 ms GPU (mx4-genesis)
and 21.7 ms GPU (mx4-devnet-epoch0) for 1 GiB; Apple OpenCL 55 and 56 ms wall. The readwidth entry's v2 figure on
this card is 0.6 to 2.1 ms cache fill and a 1 GiB build the bench-log's memory-hard entry puts at 13 to 30 ms; the
x4 build on the GPU is the row the measure lock will settle.
### 6.2a The packs as pinned after the x8 decision (`with-lock.sh run`, 22:12 UTC, the fixed binary's exports)
| Pack | How exported | Program id | Unit 0 lane 0 | Fingerprint 2^24 (Metal = Apple OpenCL) | Vectors, self-tests |
|---|---|---|---|---|---|
| mx8-genesis | `--program-class v3` (generator 3, V3_CLASS = mx8) | e323b9dcaf283a6f | 19b56348bc85304d | 7c28cfb06c5c65a9 | 3/3 + 3/3, 96 of 96, PASS |
| mx8-devnet-epoch0 | the chain path, `--program-class v3 --era-hex <genesis>` (the era drawn inside the class, load class `mx8-erad810f22d`) | 73bcbfe8ccf988f1 | d424577fce4a7a60 | 90f794dd556f7a3b | 3/3 + 3/3, 96 of 96, PASS |
| mx4-genesis (the x4 record) | `--class mx4` (generator 2, the class in the id) | 951b89584750bd75 | 63acd2d273f475ba | 6f48d5a2aa0dbe5f | 3/3 + 3/3, 96 of 96, PASS |
| mx4-devnet-epoch0 (the x4 record) | `--class mx4` on the chain seeds | (program.json) | 212c6442b51e87ae | 73caaebb28e808fe | 3/3 + 3/3, 96 of 96, PASS |
The PC 1 rows of 6.5 ran the earlier mx8-devnet-epoch0 (the load-class export, fingerprint bbb183f72692f840, era
not inside); the composed pack's PC fingerprints are owed with the next PC round (section 9).
### 6.3 Soundness suite on the v3 construction (commit 66eeba3 and after; `with-lock.sh build` for cargo, `run` for the GPU)
| Suite | Command | Result |
|---|---|---|
| Crate lib tests (memhard schedule table, the by-hand multiplied mixer, the seam, the generator) | `cargo test -j4 --release` | 44 of 44 pass |
| Pinned packs, v2 and v3 (`tests/packs.rs`: programs, ids, dataset words, 96 vectors per pack, every emitted file byte for byte, 16 masked loads per kernel, the v3 packs as the v2 seeds under mixer x4) | same | 12 of 12 pass |
| Scratch soundness tests of branch ca2-soundness (cherry-pick 0d8f745, one conflict in packbench.swift's RESULT line resolved by hand, `era_bytes: None` added to the edge program literal) | `cargo test -j4 --release --test scratch` | 7 of 7 pass: bijections, re-hit rates, 56 of 56 edge units, 42 of 42 emitted scr kernels, 200 scratch programs and 800 units on the CPU |
| v3 fuzz on the CPU (`tests/mixer.rs`): 200 programs through the seam, the contract on every instruction (each is the v2 program of its seed), 4 units each across the 32-bit range with one unit in the top 256 nonces, interpreted twice | `IGNEUM_MIXER_PACKS_OUT=<dir> cargo test -j4 --release --test mixer` | 200 of 200, 800 of 800 units; 200 packs written for the GPU runs (82 s with the three cache fills) |
| v3 stats beside v2 (`tests/mixer.rs`): 8,192 outputs per seed, bit balance, single-bit avalanche within the unit and across units, duplicates | same | igneum-genesis v3: avalanche 49.99 percent, worst bit z 1.92, 0 duplicates (v2: 49.87, z 2.25); igneum-genesis/stats1 v3: 49.97, z 3.09 (v2: 49.98, z 2.30) |
| v3 edge (`tests/mixer.rs`): items 0, 1, 2^28 - 1 and 2^32 - 1 by hand at m = 1, 2, 4, 8 on a 2^14-word cache; words 0, 15, 16, 17, MASK - 1, MASK through the interpreter's fetch path; the index wrap at MASK + 1 | same | pass |
| v3 determinism (`tests/mixer.rs`): two independent epochs, every vector and every emitted file equal, and equal to the pinned pack | same | pass |
| Metal and Apple OpenCL on the two pinned v3 packs | section 6.2 | 3/3 standalone, 3/3 in batch, 96 of 96 lanes, dataset head, MASK word and 64 samples, one fingerprint per pack across both harnesses |
| Metal fuzz: the 200 packs, 4 units each standalone and the top-256 unit inside a 512-nonce batch at base 4,294,967,040; every tenth pack on Apple OpenCL as well | `packbench --pack <dir> --batches 1 --batch-log2 9 --batch-base 4294967040` (Metal), `igneum-bench-cl-igneum-genesis-mh --bench-pack --pack <dir> --batches 1 --batch-log2 10` (Apple OpenCL), `with-lock.sh run`, 21:22 to 21:24 UTC | Metal 200 of 200 packs PASS (800 of 800 standalone units, 200 of 200 inside the wrapping window, cache and dataset self-tests on every pack); Apple OpenCL 20 of 20 packs PASS (the three vectors.h units, the self-tests); a first run with a packbench built before the `--batch-base` cherry-pick reported 200 of 200 FAIL on an empty RESULT line and was read as such (the watcher rule), the harness rebuilt and the run repeated |
### 6.4 Timings (`with-lock.sh measure`, one session, 21:40:12 to 21:40:23 UTC, commit 504cae4)
Script `measure-v3.sh` (session scratchpad): `igneum-pow bench --seed igneum-genesis --day 2026-10-03 --warps 50`
(v2), `... --program-class v3` (x4), `... --class mx8` (x8), two rounds each, then the devnet seeds, then
`packbench --pack <dir> --batches 2 --batch-log2 22 --group 256` on igneum-genesis-mh, mx4-genesis and mx8-genesis,
two rounds. The lock was exclusive among the agents' builds and measurements, but the box was not quiet: load
average 5.6 (one minute) and 26 (fifteen minutes) at the start, from unlocked processes (the devnet node, other
agents' editors); the v2 row reads 1.31 to 1.36 ms where the quiet readwidth night read 0.604 to 0.626. So the
absolute numbers below are a loaded-core figure, about 2.2x the quiet one, and the ratios between the rows are the
measurement (two rounds within 4 percent). A quiet-box re-run is owed (section 9).
| Construction | Verifier, ms per 32-lane unit, avg of 50 (round 1 / round 2) | Worst cold unit of three | Against v2 | 256 MiB cache fill, one core | Metal 1 GiB dataset build, GPU ms (round 1 / round 2) |
|---|---|---|---|---|---|
| v2 (igneum-genesis) | 1.361 / 1.310 | 1.579 | 1 | 172.1 / 172.6 ms | 29.7 / 21.0 |
| x4 (mx4, class v3) | 1.956 / 1.923 | 2.043 | 1.45x | 175.3 / 172.3 ms | 20.9 / 21.0 |
| x8 (mx8) | 2.785 / 2.790 | 2.942 | 2.09x | 172.3 / 172.3 ms | 21.9 / 21.9 |
| x4, the devnet seeds (mx4-devnet-epoch0) | 1.923 | 2.012 | | 173.9 ms | |
| x8, the devnet seeds | 2.972 | 2.885 | | 173.6 ms | |
Reading. The verifier's ALU part is what grows: x4 adds 0.6 ms per unit for 27 more mixer applications on each of
4,096 items (110,592 applications, about 5.5 ns each on this core, the lanes' chains interleaved), x8 another
0.85 ms for 36 more; the latency part (8 dependent misses per item) is the same in every row, which is why the
measured ratios are 1.45x and 2.1x and not the 4x and 8x of the M16 table's scaling. Shape B (32 rounds of one
read, section 1) would have multiplied the latency part too; x4 is under the 4.8 ms bar even on the loaded core, so
B stays unimplemented. The cache fill does not depend on the mixer (it is the ChaCha chain): 172 to 175 ms, the
spec's 175 to 181 ms of 1.8.3. The Metal 1 GiB build does not move with the mixer at all (21 ms at v2, x4 and x8
once warm; the 29.7 ms first v2 run is the first-touch cost the hosts fill twice for): on this card the build is
bound by the 8 dependent cache-line reads per item, not by the arithmetic, so the Mac says nothing about whether
the 5090's or the 9070 XT's build is arithmetic-bound; that is the PC job (section 8).
Against the x4 / x8 rule (section 6.5): the verifier half passes for x8 with 7.1 ms of the 10 ms gate to spare on
this loaded core (worst cold 2.94 ms; the quiet-core figure would be about 1.3 ms, scaled by the 2.2x of the v2
row, approximate); x4 leaves 8.0 ms. The build half waits on the PC rows.
### 6.5 Verification throughput per tier (consequences row C19), from the loaded-core figures above
Warps verified per second on one core = 1,000 / (ms per warp); a pool core verifying members' shares handles that
many shares per second; a node verifies a block with one unit (plus the header path, under 0.1 ms, not measured
here); IBD over the 108,000-header pruning window (spec 02) on one core = 108,000 x ms per warp.
| Figure | v2 | x4 | x8 | Note |
|---|---|---|---|---|
| ms per warp, steady (this session, loaded core) | 1.33 | 1.94 | 2.79 | avg of the two rounds |
| ms per warp, quiet M5 Max core (scaled by 0.604 / 1.33 = 0.45, approximate) | 0.60 | 0.88 | 1.26 | the readwidth night's v2 figure is measured; x4 and x8 scaled |
| ms per warp, 2019-class laptop core (approximate: 2.5x the quiet M5 Max figure, the ratio the design document assumes for the gate; unmeasured, O-1.14) | 1.5 | 2.2 | 3.2 | the figure that fixes the gate is a measurement, not this row |
| Shares per second per core (loaded / quiet, approximate) | 750 / 1,660 | 515 / 1,140 | 358 / 790 | spec 09 section 9.8 item 5 carried 2,270 at v2; re-cut from the quiet row: 1,660 |
| Cores for a 22,000-member pool at one share per member per 10 s (2,200 shares per second), loaded / quiet | 2.9 / 1.3 | 4.3 / 1.9 | 6.1 / 2.8 | |
| Node: worst cold single unit (loaded core) | 1.58 ms | 2.04 ms | 2.94 ms | per block |
| IBD over 108,000 headers on one core, loaded / quiet, minutes | 2.4 / 1.1 | 3.5 / 1.6 | 5.0 / 2.3 | laptop (approximate): 2.7 / 4.0 / 5.8 min; a seed VM core (unmeasured) sits between the laptop and the quiet M5 Max |
| Margin left under the 10 ms gate for Counter ASIC 3.0 (worst cold, loaded core) | 8.4 ms | 8.0 ms | 7.1 ms | on the 2019-class laptop row (approximate) 7.5 / 6.8 / 5.9 ms steady |
Reading: at x4 a pool core serves about 1,100 shares per second on a quiet 2026 core (a 22,000-member pool needs
two cores); at x8 about 800 (three cores). A node's block verification stays a few milliseconds. The gate's
remaining margin is what Counter ASIC 3.0 has to spend, and on the unmeasured laptop core it is 6 to 7 ms at x4 and
about 6 at x8, which is the number the 2019-class measurement (O-1.14) must confirm before x8 is final.
### 6.4a The same session on the fixed binary (section 6.6), `with-lock.sh measure`, 22:06:59 to 22:07:06 UTC
The verifier of section 6.4 was measured on a binary that carried the inlining regression of section 6.6; after the
fix, readwidth's binary and the fixed one on the same v2 input in the same minute (checksum 19297e99c7b9a55e), then
x4 and x8 on the fixed binary; load average 5.5 (the same box, so the ratios of 6.4 stand and the absolute row is
now the measured one):
| Construction | Verifier, ms per unit, avg of 50 (round 1 / round 2) | Worst cold unit of three | Against v2 |
|---|---|---|---|
| v2, readwidth e752fc7's binary | 0.607 / 0.610 | 0.666 | 1 |
| v2, the fixed binary | 0.609 / 0.611 | 0.657 | 1.00 |
| x4 (mx4) | 1.238 / 1.237 | 1.396 | 2.03x |
| x8 (mx8, class v3) | 2.077 / 2.058 | 2.145 | 3.4x |
| x4, the devnet seeds | 1.240 | 1.289 | |
| x8, the devnet seeds | 2.058 | 2.144 | |
The added cost per unit is the same as on the slow binary (x4 + 0.63 ms, x8 + 1.46 ms: the regression was a
constant 0.72 ms per unit in the shared item loop), so the ALU reading of 6.4 holds; the ratios against v2 are
2.0x and 3.4x once v2 is back at 0.61. The 10 ms gate keeps 7.9 ms at x8 (worst cold 2.15 ms) on this core.
### 6.5 The daily build per tier, and the x4 / x8 rule
The coordinator's rule (21:30 UTC): x8 enters v3 if the per-warp verify stays under 10 ms on one Mac core AND the
daily 1 GiB build stays under 1 s on every discrete card we own; else x4 with the thin margin stated and x8 named
as the next lever. The integrated tier is decided beside it (consequences row C23): its build is per prepare, not
per day, so its consequence is per-day dataset reuse in the workers or a restart per epoch.
| Card | Build at x1 | x4 | x8 | Source |
|---|---|---|---|---|
| RTX 5090 (PC 1, job run-mixer-x4-pc1-20261005, 22:00 to 22:04 UTC, the worker's `cache ... dataset ... ms` wall line, two dispatches per pack) | 23 to 25 ms | 23 to 25 ms | 23 ms | the PC 1 job (13.4 ms GPU time on 3 October: the wall line carries the launch) |
| RX 9070 XT (PC 1, gfx1201 on the eGPU, the same job) | 74 ms | 73 to 77 ms | 72 to 76 ms | the PC 1 job |
| M5 Max, Metal | 13 to 30 ms (the two runs of the memory-hard entry; tonight's run-lock figures 21.7 to 30.2 ms at x4 and 22.0 to 30.0 at x8 say the Mac's build is latency-bound, not mixer-bound) | section 6.4 | section 6.4 | this file |
| Radeon integrated gfx1036 (PC 2), OpenCL, per prepare | 6.9 / 9.4 / 11.7 s prepare total with the 1 GiB build inside | about 28 to 47 s (approximate: scaled x4; the iGPU's build is arithmetic-bound at x1 already) | about 55 to 94 s (approximate) | `docs/plans/epoch-length.md` section 7 (branch ca2-epoch), M11 table |
| gfx1036 beside WSL build jobs (PC 1) | 55 / 116 / 124 s | about 4 to 8 min (approximate) | about 7 to 17 min (approximate) | same |
| 8 GB-class discrete card (not owned; about a tenth of the 5090's rate, approximate) | about 0.13 s | about 0.5 s | about 1 s, on the edge of the rule | scaled from the 5090 row, approximate |
Decision (coordinator under the delegated rule, 22:05 UTC, on these rows): x8 enters class v3. Both halves pass:
the verifier at x8 is 2.1 ms per unit on one M5 Max core (6.4a) against the 10 ms gate, and the daily 1 GiB build
does not move with the mixer on any discrete card we own (5090 23 to 25 ms, 9070 XT 72 to 77 ms, M5 Max 21 ms at
x1, x4 and x8: latency-bound), 13x to 40x under the 1 s bar. `V3_CLASS = { era: None, hot: None, ..LoadClass::MX8 }`;
the pinned v3 packs are mx8-genesis and mx8-devnet-epoch0 (the latter through the chain path with the era inside the
class); the x4 packs stay pinned as the candidate's record (generator 2, the class in the id).
Reading of the integrated tier: on the discrete cards the rule is settled by the 5090 and 9070 XT rows above. The integrated tier misses the rule at x4 already: a per-prepare build of 28 to 47 s is a tenth to a quarter
of the 600-DAA-second lead the devnet gives the next program (spec 1.12), and under load it is the whole lead; so
if x4 or x8 goes in, the iGPU tier needs the workers to build the day's dataset once a day and keep it across
epochs (today a prepare rebuilds it: `proto-opencl/host.c` prepareTask builds the pair's cache and dataset per
prepare), or to restart per epoch. The node agent is asked whether the per-day reuse is bounded tonight
(coordinator, 21:45 UTC); until then the iGPU consequence stands as written.
### 6.6 The verifier regression of 0fc0ad1, found and fixed (5 October 2026, 21:59 to 22:07 UTC)
The era agent measured the same v2 input with two binaries in one minute: readwidth's 0.604 ms per unit, ca2-v3
HEAD's 1.33. Bisected under the measure lock (one session, four binaries, two rounds, checksum 19297e99c7b9a55e):
readwidth e752fc7 0.607 / 0.609; the ca2-v3 seam 6c75dad (before this branch) 0.610 / 0.609; this branch's 0fc0ad1
1.332 / 1.316; ca2-v3 88dafbc 1.325 / 1.347. So the 2.2x was in 0fc0ad1's `derive_items`, on the version 2 path the
devnet verifies with, and section 6.4's "loaded box" reading was wrong: the load was real (the same session shows
it) but the 2x was the code.
Variants, each a one-change copy measured against readwidth's binary in the same session:
| Variant | ms per unit | Reading |
|---|---|---|
| 0fc0ad1 as written (the line mask read from the cache at run time, the loop inlined into `MemhardCpu::fetch`) | 1.33 | the regression |
| the mask hoisted into a local before the item loop | 1.32 to 1.37 | not the reload |
| `Cache::line` with the constant mask, on the 0fc0ad1 tree | 0.60 to 0.65 | fixed there |
| the same constant mask on the merged ca2-v3 tree | 1.33 to 1.46 | not the mask either |
| one instance per cache size with the mask a constant, `#[inline(always)]` | 1.32 to 1.39 | not the mask |
| the same instances `#[inline(never)]` | 0.604 / 0.617 / 0.618 / 0.624 | the fix |
So it is inlining: the item loop inlined into its callers (`fetch`, `fetch_wide`, `word_at`) runs at 2.2x the
out-of-line loop, and which small change tips LLVM's decision depends on the rest of the tree (the constant mask
tipped it on one tree and not on the other). The fix (`memhard::derive_items`): the loop is `derive_items_mask`,
`#[inline(never)]`, one instance per cache size the growth rule reaches (2^26 to 2^30 words) with the line mask a
constant, and a run-time-mask instance for every other size (tests). Measured in 6.4a: 0.609 / 0.611 against
readwidth's 0.607 / 0.610.
The class, not the instance: a verifier benchmark with a pinned bound in the crate's CI (the v2 unit at a known
input against a stored ms-per-unit on a named core, failing on a 1.3x drift) would have caught this at the first
commit; filed for the next cut (section 9). Until then the era agent's two-binary check (same input, same minute)
is the rule for every change that touches the item loop.
## 7. The chip model
`docs/analysis/chip-model-v3.md`.
## 8. What is unverified
1. The 5090's and the 9070 XT's dataset build at x4 and x8 are measured (6.5, the PC 1 job): both latency-bound,
under 0.1 s. The gfx1036 is the tier that fails the per-prepare build (6.5), and its consequence (per-day dataset
reuse in the workers) is with the node agent.
2. The absolute verifier figures were taken on a loaded core (load average 5.6); the ratios are the measurement
and the quiet-core figures are scaled. A 2019-class laptop core has not run any construction (O-1.14).
3. The mixer has had no cryptanalysis (spec 1.8.4); `m` applications with distinct round keys is `m` times the
work only if no shortcut composes them, which is the same open question as for one application.
4. The x8 packs are generator 2 with the class in the id (`--class mx8`); if x8 is chosen, the pinned v3 packs are
re-cut through the seam (`V3_CLASS = MX8`, generator 3) and the tests re-pinned, one commit.
5. The 5090's rate for the v3 program is the v2 rate by construction (the hash kernel is unchanged, the Mac shows
27.7 MH/s at v2, x4 and x8); the chip row's denominator stays the readwidth table's 136.1 MH/s until a v3 pack
runs on the card, which the PC job also gives.
## 9. Owed
| Item | Owner | When |
|---|---|---|
| PC 1 run of the five packs | done 22:04 UTC (run-mixer-x4-pc1-20261005; 6.5 and 6.2a) | |
| A verifier benchmark with a pinned bound in the crate's CI (section 6.6, the class rule) | ca2-mixer | the next cut |
| GPU bit-exactness of the re-exported mx8-devnet-epoch0 (the composed class, the era inside) and the mx4 record packs on the PCs (the Mac rows are in 6.2a) | ca2-mixer | the next PC round |
| The x4 / x8 choice recorded from the rule, then the vectors re-cut once through the seam | done 22:05 UTC: x8 (6.5), V3_CLASS = MX8, mx8 packs re-exported | |
| `Epoch::chain_dataset_day` wired to the genesis day index in the node (`days_since_genesis(day_index(header), day_index(genesis))`) and `pow_genesis_dataset_log2` in the override | ca2-node | the integration |
| The spec text of section 2 into `docs/spec/01-lottery-hash.md` 1.8.5 and 1.13.3 (with the v3 vectors into 1.17) | the integration | after the choice |
| The 2019-class laptop core measurement that fixes the gate (O-1.14) | cryptographer | gate 1 |

View file

@ -0,0 +1,84 @@
# Morning summary, 6 October 2026
Written for the project lead at 08:40 UTC. Every number is in `docs/bench-log.md` or the named plan with the command that produced it. Failures are listed with the passes.
## The headline
Everything on the overnight list landed. The one thing that could have gone wrong, switching the live chain to program class v3, held.
| Piece | State | Number |
|---|---|---|
| 0.3.11 (class v3 + proving v1) | On master 630da6b, three master CI runs green, rolled out to the Mac, PC 2, the seed, both hand nodes; PC 1 took it at 07:00 | Live manifest 0.3.11, nine-field override, digest 0139ab9d |
| Counter ASIC 2.0 crossing | Crossed at 03:51:42 UTC, watcher verdict PASS | 58.7 blocks a minute before, 59.2 after; three hourly swaps since, no pause, 0 refusals |
| 12 GB proving | Floor broken on a patched SP1 server (`prover-floor` bcc6d68); real card lands today, test on PC 2 | Proves alone: 10.3 GB, 5.7 s a shard. Mines AND proves as a core-only prover handing its proof to a big-card aggregator: 8.1 GB beside the miner, 19.8 s a shard (measured 09:15 on the 5090's allocation). 16 GB mines and proves compressed (12.9 GB, 17.4 s). 8 GB stays out |
| On-die recompute chip | Emulated on the 5090's own L2 (`ledger-pc2` 564acab) | 0.256x honest, 5.1x worse per joule |
| Ledger | Rounds 2 and 3 closed (`fud-close` d16bc3b, `ledger-rebase` abb08a5) | 47 consequence rows, 42 closed or taken, 14 decisions |
| Branches ready for later cuts | pool-v0, rig-install, repro-bench b776199, ember-tune 9a6469f, ota-k2, asic-history, proving-methods | measured where PC 2 allowed |
## Since you got up (07:00 to 08:40 UTC)
| What | State |
|---|---|
| PC 1 | Back on 0.3.11 at 07:00, 5090 and 9070 XT mining; integrated card off. The update needed your click because the app's hourly rollout slot had not come; fixed as a catch-up rule (`update-catchup` 2207cd7, 0.3.12) |
| Mac | Mining paused through the app (persists); its node runs |
| Chain | 19 miner ids, 225 MH/s on the two PCs |
| Zero proven segments | Solved: the prover claims a whole segment and proves its 8 blocks in order (9 segments in 30 minutes on PC 2 beside the miner, 72 of 72 shard records paid, 11 percent hash cost). The segment record itself needs a consensus switch on the node fork (`proving_v1_fresh_rule_daa`), folded into 0.3.12 |
| Ember Tune re-run on PC 1 | Failed at 07:56 with no rows: the playbook wrote the test engine's settings with a byte-order mark, the engine parsed defaults, sat idle 35 minutes. Nothing was set on either card; the app restored its miners by itself. Fixed (8273494, watchdog 1e9550e, CI check). Re-run `ember-tune-pc1-3` needs one more click when you are back |
| GPU list order | Cards ordered by performance, integrated last (`card-order` ffb2bfa, 0.3.12) |
| Counter ASIC 3.0 | Running, all seven items plus a new item 8. See below |
| 0.3.12 | Being prepared: the app items plus the node fork with the segment switch; stops at the publish gate for your go |
## Counter ASIC 3.0 so far
The finding that matters: the chip that wins is not the clever recompute chip 2.0 priced. It is a stored-dataset chip, the whole dataset in DRAM, a 28 nm memory-controller die doing dependent reads.
| Attacker | Per chip vs 5090 | Per joule vs 5090 | Source |
|---|---|---|---|
| On-die recompute chip (2.0's model) | 0.92x with the 3x allowance | 1.86x | chip-model-v3 |
| Same, emulated on the 5090's L2 | 0.256x | 0.2x | M16 inline bench |
| Stored-dataset chip, GDDR7 | 1.22x | 5.1x | item 1 |
| Stored-dataset chip, HBM3 | about 1.2x | 7.5x to 9.2x | item 1 |
| Ethash precedent (E3, A10) | | 2.1x to 4.8x | history rows 3, 4 |
| Stored-dataset chip with the latency shadow filled (item 8, N = 100,000, parity cores) | | 2.7x vs 5090, 1.4x vs M5 Max | item 8 Mac rows |
Why: at the hash the 5090 spends about 55 W on memory and the rest keeping a GPU alive at 0.15 percent of its integer budget. The lever is RandomX's lever: make the hash use the rest of the chip. The 5090 can hide about 330,000 operations per hash behind its 128 reads; today it hides 512. Item 8 measures that fill: on the Mac it costs 1.5 percent of rate at 100,000 ops, the verifier barely notices, and the chip's edge drops from 5x to 2.7x. The 5090's rows are queued on PC 2. The deciding number is the chip core's energy per op against a GPU's ALU, which item 8 is pricing.
**Closed at 08:50.** One class v4 candidate, measured on the hash's own numbers and ready for its six-gate run on your word: mixer x8 plus 100,000 operations of program work per hash. The 5090 loses 0.2 percent of rate and the M5 Max 1.5 percent; the verifier adds 0.17 ms per warp; bit-exact on Metal, CUDA, Apple OpenCL and the CPU emulation. The chip must then carry a 14,000-lane ALU array, and its per-joule edge over the 5090 falls from 5.6x to 2.1x at a core as efficient as the GPU's, 1.5x at a realistic one. Your test as a number: the chip crosses 2x only if its datapath spends under half the energy per op that a GPU does. The cost per tier: a 5090 draws 431 W instead of 350 for the same blocks (a rig pays about 23 percent more electricity), the M5 Max 37 W instead of 21, a pool user sees nothing. The 9070 XT and 4060-class rows are owed, the AMD ones because PC 1 was left alone.
Other items: item 2 (per-day random derivation) works bit-exact at no hash cost and drops the recompute chip to 0.29x to 0.43x, but at full size its CPU verifier is over the 10 ms gate on an old core; it goes in as a reserve, the half-size draw passes, and the class v4 candidate is the pairing of derivation class and program length under one verifier gate. Item 3: cryptanalysis brief and budget line, USD 80,000 to 160,000, one firm plus one academic group, verdict GO to commission, nobody contacted. Items 4 and 5: share-pattern detector in the observer (fires on a fabricated fixed design, quiet on the devnet), issuance trigger at USD 20,000 a day, FPGA soft overlay 0.3x to 0.4x a 5090 per watt, layer 9 ranked above layer 7. Items 6 and 7 in flight.
## Decisions for you
The full list with recommendations is in `consequences-decisions.md` (14) and the 3.0 status file. The ones that bite first:
0. **Run the six gates on the class v4 candidate**, or wait for the 9070 XT and 4060-class rows first (3.0 status, decision 2).
1. **The public claim "under 2x".** True of the recompute chip per chip, false of the stored-dataset chip per joule. Two re-wordings are drafted in the 3.0 status file; nothing on the site changed. Pick one before any public push.
2. **The segment rule.** Answered at 08:50: it is a consensus rule (as shipped a fresh segment record is valid for an 8-second window). The fix is on the node fork behind a new switch `proving_v1_fresh_rule_daa`, so 0.3.12 carries the node and goes out as a two-manifest publish with the switch at tip + 14,400. Measured on PC 2 beside the miner: 9 whole segments in 30 minutes, 72 of 72 shard records paid, 11 percent of hash rate.
3. **0.3.12 go.** The state-reply fix, the update catch-up, the card order, Ember Tune and its guards, plus the node fork with the segment switch (two-manifest publish, switch at tip + 14,400). No prompt on any machine.
4. **The 12 GB card.** Into PC 2 when it lands (PC 1 has no prover toolchain); the playbooks are written. Core-only provers need a pool-protocol change (a core hand-off format only aggregators accept, the compressor's credit, the prover's signature over the proof hash); nothing on chain changes. Decide whether that goes into the pool spec now.
5. **Cryptanalysis budget** (D11 and item 3), **growth mapping (b)** (D4), **a release-tag convention**, **the bounty only once escrowed**.
## What went wrong, plainly
- Epoch-34 pack outage, 18:23 UTC, 56 minutes of both PCs down: a pack-attempt bug in the worker loader. Hot fix shipped by job; class fix in 0.3.10.
- GitHub Actions outage forced a PC-built 0.3.10. 0.3.11's first CI run then failed in the census crate nobody updated for class v3; fixed (2a62735), rerun green.
- Credits ran out at 19:58 UTC and killed three agents; resumed.
- The 9070 XT dropped off the bus four times; the reading was wrong, the relay's "PC1" is PC 2.
- A 2.2x verifier regression on the mixer branch, caught before publish.
- Two Mac-only v3 outages caught by the Metal gate before publish.
- "12 GB proves" was false on the shipped prover; the floor was SP1's server code; patched.
- A job quit PC 1's app at 22:31 UTC (a second engine ran the urgent updater, whose installer sent quit). Rule written, CI gate added; PC 1 stayed down all night. Fixed in e600e63.
- Three dark proving windows on PC 2, all the same state-reply bug (paid_wei above u64::MAX empties the reply); 0.3.12's first item.
- The Mac miner lost its node subscription for 11 minutes after a node restart (C43, 0.3.12).
- The 0.3.11 update on PC 1 waited on the hourly slot and needed your click; catch-up rule in 0.3.12.
- Ember's first real run burned 35 minutes and your prompt on a byte-order mark; fixed with a watchdog and a CI check.
- The 07:45 summary task never fired on its own and its manual run stalled on a tool prompt; this document was written by hand.
## Today's hands list
1. 0.3.12 go when the shipper reports it green.
2. One click for the Ember re-run on PC 1.
3. The 3060 into PC 2; the playbook runs the same fixture as the sweeps.
4. 16:00 UTC: the fee-switch check (every prover on 0.3.11 before H = 210,000 at about 19:15 UTC).
5. USB copies of the key backup (10:00 reminder).
6. The decisions above.

223
docs/plans/proving-v1.md Normal file
View file

@ -0,0 +1,223 @@
# Proving v1: segment records, the chain rule, the unproven rule; the 0.3.11 rollout
5 October 2026, from 18:55 UTC (the project lead: "open the proving round asap"). Branches `proving-v1` in the main repository
(worktree `/Users/joshm/Projects/igneum-wt-proving-v1`, from master a93199a) and in the fork
(`vendor/igneum-node-pv1`, from release-0.3.6 a24ab01a; to be rebased onto the 0.3.10 tip when it lands on
release-0.3.6). Status words follow `docs/spec/00-overview.md` 0.2. Every number here is in `docs/bench-log.md`
with its command. Nothing ships from this plan: it delivers branches, numbers and the rollout for 0.3.11.
## The gap this closes
The litepaper says every block is proven within about a minute. Proving v0 (`proving-v0.md`, spec 7.7) proves
some shards: one prover (PC 2's RTX 5090) takes the newest shard assigned to it, about one shard every 30 s, so
under a tenth of blocks carry a proof; consensus does not require one; the aggregator guest (design 5.3) runs
on fixtures only. Proving v1 adds the aggregated segment record on chain (spec 7.8), the chain rule (segment N's
record verifies N-1, inside the proof by recursion), the unproven rule (a segment nobody proves in T seconds pays
nothing and may be skipped), the prover on by default on every machine that can prove, and the measurements
that say how many cards cover the chain.
## The round, step by step
| Step | What | State |
|---|---|---|
| 1 | Prover on by default (`app/igneum-app/src/provedefault.rs`, `engine.rs apply_prove_default`): on at install when the machine can prove (NVIDIA card with 12 GB or more; WSL2 answering on Windows; Linux native; Apple silicon off until measured), never switching an explicit on back off; the Settings switch line and the tile line say why. Unit tests (5). The cost of proving on a mining machine: PC 2 job `prover-cost-pc2-pv1` (5 min mining alone, 5 min with the prover, the GPU memory peak and the host RAM peak, the sp1-gpu-server's compiled SM targets) | Implemented; the measurement is HELD (coordinator, 19:00Z): PC 2's RTX 5090 worker has been exiting on a pack seed mismatch since 18:35Z, so the first run's "mining alone" is 0 MH/s and void; re-run after the go |
| 2 | Segment aggregation: `SegmentRecord` (586 bytes, the aggregator guest's 340-byte statement inline), section `IGNS` before the shard section, p2p message 75 at protocol 15 (14 went to the EVM transaction relay in 0.3.10), the native block statement and the veto, the credit split (`split_pool_credit`), the payout at the carrier, `igneum-miner sign-segment-record`, the RPCs; host modes `chain` (consecutive fixtures), `aggregate` (live shard proofs from the pool, a run of blocks in one process) and `verify-segment` (the node's verifier, pinned aggregator key); the app's aggregator step (`prover.rs aggregate_once`) | Implemented, unit-tested (consensus core 2 new tests, exec 2, params 1); the GPU measurement (N = 2, 4, 8 blocks on PC 2) is HELD with step 1; the Mac CPU run of `--mode chain` over 2 live blocks is the known-finished case |
| 3 | Coverage: `tools/proving-v1/coverage.mjs` (the proven-block share and the on-chain proof latency over a window from one node's RPC, the live page beside it) | Implemented and run for 3 min (below); the 30-min window waits for the fleet |
| 4 | The chain rule and the unproven rule in consensus behind `proving_v1_activation_daa` (spec 7.8 items 2, 6, 7); unit tests; the fast-time 3-node harness `tools/proving-v1/net.mjs` (ports 29950+, suffix 956, trust mode) with the known-finished and known-failed cases | Implemented; the harness run waits for the Mac build of the fork (`vendor/igneum-node/target-pv1`) |
| 5 | This plan: the rollout for 0.3.11 and the project lead's decisions | Written below |
## Numbers (every one from `docs/bench-log.md`, "proving v1: segment records ...", 5 October 2026 evening)
| What | Number |
|---|---|
| The prover's cost to a mining 5090 (the re-run with the fleet mining, hash rate from the miner's own STATUS lines) | 124.7 MH/s alone, 119.7 MH/s with the prover on: 5.0 MH/s, 4.0%, on empty shards at 1.4 a minute |
| GPU memory on the 5090: the prover alone (empty shards) / the miner and the prover together | max 13,816 MiB / max 15,590 MiB (the miner holds 3,396 MiB); a 24 GB 4090 has 8.4 GB of headroom, a 16 GB card 0.4 GB, a 12 GB card cannot do both on this build; the full-shard peak is the chain job's row |
| Coverage, 30-min window with the fleet mining, one prover | 2.4% of blocks, latency p50 44 s, p99 52 s |
| Host RAM | host used 25.6 GB of 63 GB; the WSL2 VM 7.9 GB working set |
| sp1-gpu-server 6.8.1 compiled targets (cuobjdump) | sm_80, sm_86, sm_89, sm_90, sm_100, sm_120 and compute_120 PTX: Ada (4090) is native, no JIT; nothing for AMD |
| Shards a minute, one 5090 through the app's loop (empty shards) | 1.6 |
| Chain of 2 live blocks on the Mac CPU (`--mode chain`) | shard 55.4 and 41.3 s, aggregate 52.0 s then 59.1 s with the previous proof, chain_len 2, final proof 1,272,909 bytes, `verify-segment` 0.032 s |
| Unit tests | consensus core 13, exec 8, app 5, all passing on the Mac |
| The harness (3 nodes, fast time, trust mode) | PASSED, 21 checks in 197 s on b177718e (N = 4) and 244 s on the final tree ece42979 (N = 8): paid 1.0 s after submit, every node agreeing; the fresh chain refused after a proven segment; the unproven segment skipped after its deadline; shards at 90% |
| Coverage, 3-min window, the degraded fleet (one card, the Mac verifier down) | 4.7% of blocks proven, on-chain latency p50 39 s |
| The chain of 8 live blocks on the 5090, the card also mining (`chain-pc2-pv1c`) | shard 7.3 to 7.7 s, first aggregation 7.9 s, every chained one 9.6 to 9.7 s; N = 2 in 32.6 s, N = 4 in 66.8 s, N = 8 in 135.6 s (17.0 s a block); the final proof 1,272,909 bytes whatever N, the record 586 bytes, `verify-segment` 0.037 to 0.040 s; GPU peak 16,751 MiB with the miner resident. Against 4 October with the miner stopped (aggregate 2.2 s): the miner slows the prover 3 to 4x |
| 5090-class cards for 100% at 1 block/s, measured rows | 47 with the loop as it is, 18 through the chain mode on mining cards, 6 (approximate) on proving-only cards, at empty blocks; 14 proving-only at one full shard a block; 45 at `B_p`: the table in the bench log |
## The rule, in one paragraph (spec 7.8)
From the first chain block `A` at or above `proving_v1_activation_daa`, chain blocks form fixed segments of `N`
(`proving_v1_segment_blocks`). A segment record carries the aggregated proof of the segment's last block, whose
`chain_len` says how many consecutive blocks the recursion attests. Every node checks the record's statement
against its own native block statement (every field but the provers commitment and `chain_len`), the chain rule
(a proof that does not chain to the previous segment, `chain_len = N`, is valid only for the first segment or
after an unproven one) and the deadline (`T = proving_v1_unproven_daa` DAA seconds after the segment's last block;
a record carried later pays nothing). The pool credit of every attested block splits: `proving_v1_aggregator_share_bps`
to the aggregator, the rest to the shards as v0. A block is never invalid for lack of a proof; the mandatory rule
(spec 7.8 item 10) is Designed and off, with no switch yet.
## Rollout for 0.3.11 (the digest handshake pattern of 0.3.9 and tonight's switches)
The switch moves the consensus digest only once it is set (`consensus_digest`: the four v1 fields enter the hash
when `proving_v1_activation_daa != never`), so a 0.3.11 node on the unswitched devnet keeps the 0.3.10 digest and
the rolling upgrade does not partition the network. The order, each step with its check:
1. **Rebase and build.** Fork `proving-v1` rebased onto the 0.3.10 tip on `release-0.3.6`; the six node suites and
the app tests as PC 2 build jobs; the Mac node and the Windows exes by the Mac cross-build; the Linux node by
PC 1; the HiveOS package republished from the same fork commit (rule: the HiveOS package carries the node of
the release commit and the same override object, `infra/hive`, as 0.3.9's `hive-sync-039o` checked it).
2. **Pinned guests.** The guest ids do not change in this round (shard `0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a`,
aggregator `0x474678f35f7545db28055d5e5bbc308231d84a5a072202087a2a8d5b09123896`, pinned 2026-10-05T16:20:38Z):
the aggregator guest already carried the chain rule and only the HOST gained modes. So no provers-off drain is
needed for the guests; `--mode id` on every machine after the update must print the same two ids, and the node
now reads them at start (`program_ids`: `IGNEUM_PROOF_PROGRAM_IDS` or the verifier's `--mode id`) and names them
in the native statement.
3. **Hand nodes and the seed first**, with the UNCHANGED override object (the digest stays): observer, node 1, the
seed on the 0.3.11 node; peers back within 20 s; the `proving v1: segment records from DAA score never` line in
each log.
4. **Manifest and apps.** `publish-manifest.sh --version 0.3.11` with the unchanged object; `update-now` to every
app; every machine on 0.3.11 with a DAA score and a hash rate (the watcher takes the commit as an argument).
The app's prover default applies at the first start on 0.3.11: every NVIDIA machine with WSL2 goes on; the
log line `prover default: ...` on each.
5. **The switch.** When every node runs 0.3.11: publish the object with `proving_v1_activation_daa` = H (24 h
ahead, the rule of `fee-switch-devnet.md`) and the three parameters; read the expected digest on a scratch node
first; `update-now`; the hand nodes and the seed with the same object; the digest sweep; the first paid segment
record (`igneum_getSegmentRecords`) and `igneum_getProvingStatus.v1.segmentsInWindow` after H.
6. **The mandatory rule** stays off: no switch exists for it yet; it gets one when the measured share is one.
## The decisions (Decided 5 October 2026, delegated: the project lead, "I have no idea for most of this stuff so do a lot of research and deploy what is absolute best")
Each with its rule, its number and its evidence. The deploy is the DEVNET through 0.3.11 (not the public testnet).
### What the other networks do (read 5 October 2026, 20:05 to 20:15 UTC; every figure from the page named, else labelled approximate)
| Network | Unit proven | Deadline | What a miss costs | Who is paid what | Measured latency |
|---|---|---|---|---|---|
| Taiko Alethia (L2BEAT page, protocol v2.1.0 notes) | a batch of blocks | proving window 2 h, cooldown 2 h (v2.1.0, February 2025); the Shasta inbox targets a 4-h proof submission cadence | the proposer's liveness bond is credited back in full when the batch is proved inside the window, half when outside; mainnet currently sets minBond and livenessBond to 0 | the prover earns the proving fee; two of four proofs needed (SGX Geth, SGX Reth, SP1, RISC0, at least one ZK) | 100% ZK coverage of mainnet blocks reached December 2025 (blockchain.news); preconfirmations 2 s |
| Boundless (docs.boundless.network, proof lifecycle) | one request | the requester's timeout (example 3,600 s) and a lock timeout (example 2,700 s); a reverse Dutch auction ramps the price from the minimum to the maximum over a ramp-up (example 300 s) | the locked collateral (example 5 ZKC) is slashed and used to pay another prover who fulfils the request | the prover's fee = the bid minus the market fee | not stated on the page |
| Succinct Prover Network (docs.succinct.xyz, SPN architecture and quickstart) | one request | the requester's deadline (the quickstart example: 10 minutes, 50 PROVE staked to bid, 100 PROVE maximum fee) | part or all of the winning prover's collateral slashed "according to protocol rules" | a reverse auction: the lowest bidder is assigned | "real-time", no number on the page |
| Aztec (docs.aztec.network economics; L2BEAT; forum) | an epoch of 32 blocks (38 min 24 s), a proof may cover one checkpoint (1 min 12 s) up to one epoch; maximum proof window 1 h 16 min | the epoch is declared failed only when its submission window expires | an unproven epoch is reorged out (no reward); proposals under discussion remove bonds and pay every prover that delivers on time | 400 AZTEC a slot: 70% sequencers, 30% provers (120 AZTEC), provers' share by an activity score | the public testnet proved by community provers (zkcloud blog), no page number |
| zkSync Era, Linea, Scroll (eco.com comparisons) | a batch | none on chain (the operator proves) | none | the operator | proof latency about 30 min (zkSync Era), 75 min (Linea), 90 min (Scroll), approximate |
Reading. Nobody pays an aggregator as a separate role: Aztec's 30% goes to whoever delivers the epoch proof, Taiko's fee to whoever proves the batch, the markets to the request's winner. Deadlines run from 10 minutes (Succinct's example) through 1 h (Boundless' example) to 2 h (Taiko) and 1 h 16 min (Aztec's maximum window); a miss forfeits the reward or part of a bond, and the slashed value goes to the prover who steps in (Boundless). Igneum has no bond on shards by decision (spec 7.2 item 4), so the forfeit here is the reward only.
### The decisions
| Decision | Decided | Rule and number | Evidence |
|---|---|---|---|
| `proving_v1_segment_blocks` (N) | **8** | Aggregation is a fixed cost per block, not per segment: 9.6 to 9.7 s for every chained block on a mining 5090, 7.9 s unchained (`chain-pc2-pv1c`), so N buys nothing in card time; it sets the record cadence and the forfeit. At N = 8 and 1 block/s a record every 8 s, a 1.27 MB proof gossiped every 8 s (159 KB/s per path, half of N = 4's 318 KB/s), and a missed segment forfeits 8 blocks' aggregator share (8 x 0.088 IGN at today's credit). The chain for 8 blocks cost 135.6 s cold on a mining card (66.8 s for 4), a fifth of T; pipelined per block it is 17 s after the last block. Aztec proves 32 blocks (38 min) as one; 8 blocks at 1 block/s is 8 s of chain, so the record lands well inside the minute the litepaper promises | bench-log "proving v1" chain rows; Aztec economics page |
| `proving_v1_unproven_daa` (T) | **600** DAA s (10 min) | T = p99 x 10: the measured block-to-carried-record latency of a shard record is p99 52 to 62 s (two 30-min windows), a cold chain of 8 adds 136 s and relay plus inclusion 10 to 40 s, about 240 s worst case; 600 leaves 2.5x on that and equals the 600-block record window of v0, so nothing is payable past it either way. Succinct's example deadline is the same 10 minutes; Boundless' example 1 h, Taiko 2 h, Aztec up to 1 h 16 min: Igneum's blocks are 1 s and its proofs seconds, so the shortest of the field. The forfeited aggregator share of an unproven segment STAYS IN THE POOL ESCROW (it is never paid, as an unproven shard's part today): no burn and no roll-over, the rule the pool already has, and the escrow is what later proofs are paid from | coverage rows; `chain-pc2-pv1c`; the table above |
| `proving_v1_aggregator_share_bps` | **1,000** (a tenth) | The aggregator's card time per block is 9.7 s on a mining card against 4 x 10.6 s of shard proofs at `B_p` (19% of the card time) and 2.5 s against 42.5 s with the card to itself (6%); on tonight's empty blocks it is half the card time. A tenth of every attested block's pool credit sits between the two full-block ratios, pays a role no other network pays separately (Aztec pays its 30% to whoever delivers the epoch; the markets pay the winner), and leaves the shard provers 90%, which the fast-time harness showed paid exactly (shardWei 90% of the credit). The pool's 20% emission share itself is unchanged (spec 2.5, 5.3) | `chain-pc2-pv1c`; bench-log 4 October 5090 rows; the harness |
| `proving_v1_activation_daa` (H) | **the devnet tip + 14,400 at publish** (4 h at 1 block/s), set by the 0.3.11 publisher in the same override object as `program_class_v3` | tonight's rule for consensus switches (the coordinator, 5 October 2026); the digest moves only once H is set, so the rolling update does not partition |
| Aggregator sortition | **none in v1**: the first valid record carried wins | design 5.3's VRF draw (O-7.3) with one or two aggregators on the devnet changes nothing; the segment grid and the deadline already bound the race; revisit when a second aggregator exists |
| Apple silicon default | **off** | the gate was "a shard under 60 s with the miner running": the M5 Max CPU took 41.3 and 55.4 s for EMPTY shards under tonight's load and 272 s for a 200-pgas shard on 4 October; a full shard at `S_p` was never under 60 s. Settings switches it on | bench-log "proving v1" CPU chain row; 4 October CPU rows |
| The prover profile per card and the 12 GB and 16 GB gates (the project lead: "make sure we can prove on 12gb cards"; "is there any way we can make 12gb cards mine and prove?") | **measured on PC 2, the rows below** | the SP1 6.8.1 GPU server reads `ELEMENT_THRESHOLD`, `HEIGHT_THRESHOLD`, `SHARD_SIZE` and the `SP1_WORKER_NUM_*`/`BUFFER_SIZE` knobs from the environment it inherits (`sp1-core-executor-6.8.1/src/opts.rs`, `sp1-prover-6.8.1/src/worker/config.rs`); the app passes a profile per card (`provedefault.rs`) and the host forwards it | the sweep job `memsweep-pc2-pv1` and the miner-on run |
The resume path (5 October 2026, the 0.3.11 app): `POST /api/resume` on 0.3.9 re-armed only FAULTED cards (`stop_miners("paused")` clears every slot's `restart_at`), so a healthy paused card stayed "off" at 0 MH/s until the app was relaunched: PC 2 at 21:25:11Z (the aggregation-cost job's pause and resume; `[ok] mining resumed` then `0.00 MH/s, waiting` for 20 minutes), the Mac that afternoon. Now every slot without a live worker is re-armed and its pack exported again before the start, and 90 s later `resume_check` logs `resume: <card> is not mining 90 s after resume (state ..., pid ...)` for every enabled card without a hash rate (`engine.rs`, three unit tests: the state machine, the 21:25:11Z case against the old rule, the check).
The prover-floor agent's first sweep (job `floor-sweep-1`, 22:34 to 22:38Z, PC 2's 5090, the miners stopped, this plan's per-point recipe, its patched `sp1-gpu-server` 5568108b built for sm_86, sm_89 and sm_120, every proof VERIFIED by the unpatched pv1 host): the control at upstream's sizes reproduces the curve above (empty shard 13,892 MiB and 2.2 s; the v1 shard 20,516 MiB and 4.2 s); with the core element threshold at 2^26 the v1 shard proves as four core shards in 5.3 s at **12,708 MiB** and the empty shard at 12,772 MiB; 2^25 gives 12,836 MiB at 8.5 s; 2^27 gives 15,396 MiB. The 12.7 GB left is the server's Setup (five recursion keys pre-built at a fixed 2^27 capacity plus the shrink and core keys: 9.7 GB before the first shard), which its patch v2 sizes to the need. Decided for the 12 GB profile: the split that lands under 11 GB wins (5.3 s a shard is inside the loop's own 25 to 30 s of carriage and 100x inside T); 2^27 is the second profile only if v2 leaves it under 11 GB with the miner's 1.8 GB beside it. The 12 GB row stays OPEN until the final pair (alone and beside the miner) lands and the on-order 3060 runs it.
### A self-built CUDA server (the 12 GB path), before 0.3.12 (consequences C26)
If the prover-floor agent's rebuilt `sp1-gpu-server` (the Setup sizes cut, built on PC 2 under WSL2) proves a shard under 11 GB, it becomes a shipped artefact and needs its own row of rules before 0.3.12: it is built from a pinned SP1 source tag with `CUDA_ARCHS` covering sm_86, sm_89 and sm_120 (the 12 and 16 GB tiers are Ampere and Ada, not only the 5090's Blackwell; one card family per measured row), by the packaging path that builds the Windows payload (PC 1's build job for the Linux binary, the Mac signs the manifest as it does the DMG), lands in the DMG and the WSL2 package beside the host as `wsl2/bin/sp1-gpu-server` with its sha256 in `payload-inputs.json`, is named in `evidence.md` beside the prover rows ("prover built from SP1 <tag> at <sha>"), is rebuilt and re-measured at every SP1 upgrade, and ships only after `--mode verify-segment` and `--mode verify` on proofs it made show the pinned verifying keys unchanged (the server changes allocation, not the circuit; the ids `0x2b1a81cb...` and `0x474678f3...` must still verify them). The 12 GB claim itself waits for the on-order RTX 3060 to run that server on the same fixtures and recipe as the curve; until then the public line stays at 24 GB.
The root-socket class on PC 2, the two times: 20:00:56Z (my chain job's root run; the live prover failed with Connect(PermissionDenied) until the socket was gone) and 21:25:24Z (the aggregation-cost job's root run; the prover stayed dark through the 0.3.10 restart at 21:49:41Z until `socketfix-pc2-pv1` removed the root-owned `/tmp/sp1-cuda-0.sock` at 22:01:16Z; the next shard, block 89011, was proven at 22:02:13Z and paid, and every shard since). The permanent fix in the 0.3.11 app tree: every committed playbook that runs a prove mode as root carries `pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at its start and end, `tools/ci/prover-socket-check.sh` (in `ci.yml`) fails a playbook without them, and the app's prover names the cause in its log line when the host reports PermissionDenied. The app itself cannot remove a socket another user owns, so a job written outside the tree must still follow the rule.
A prover job on a shared card runs as the app's user or cleans its socket (`pkill -f sp1-gpu-server; rm -f /tmp/sp1-cuda-*.sock` at the start and the end; `tools/ci/prover-socket-check.sh`): the root-socket fault of 20:00Z, bench-log.
### The prover profiles: the tiers from the S_p curve (bench-log, "proving v1", the sweep, the miner-on pair and the curve)
The GPU server of SP1 6.8.1 sets the memory, not the shard: a floor of 13.9 GB for an empty shard, 20.4 GB for a full shard at the adopted v1 budget (30,000 pgas, 4.7 M cycles), 28.3 GB for the prototype shard (6.75 M pgas, 60 M cycles); the miner adds 1.7 GB when it shares the card; no environment knob moves the floor and the server has no options of its own; the witness is 5 to 22 KB a shard and never binds.
| Card | Alone | Beside the miner | Default (`provedefault.rs`) |
|---|---|---|---|
| 32 GB (RTX 5090) | the prototype shard, 28.3 GB, 10.8 s; the v1 shard 20.4 GB, 4.3 s | the prototype shard 30.1 GB, 33 s; the v1 shard 22.2 GB, 13.2 s | on, mine and prove, today |
| 24 GB (RTX 4090, 3090) | the v1 shard 20.4 GB; the prototype shard does NOT fit (28.3 GB) | the v1 shard 22.2 GB measured on the 5090's allocation (2.3 GB spare on a 24 GB card; approximate for the card itself) | on, mine and prove, with the line "until the devnet's fee switch its shards are the prototype size, which needs 32 GB, so this card proves from the switch on" |
| 16 GB (RTX 5080, 4080) | the shipped server: an empty shard only (13.9 GB); the patched server v3 b37defef at threshold 2^27: the v1 shard 12,915 MiB and 4.3 s, the prototype shard 13,459 MiB and 16.8 s (measured by the prover-floor agent on the 5090's allocation, job `floor-sweep-3`, 00:13 to 00:17Z 6 October; 2^27 + 2^26 gives 16,115 MiB, over the card) | the shipped server: nothing (15.7 GB for an empty shard); the patched server at 2^27 beside the miner (the 5090 mining at 95%, 338 W, same card; `floor-sweep-4`, 00:35 to 00:38Z): the v1 shard 14,786 MiB total with the miner's 3,833 MiB resident inside it, the server's own 10,953 MiB, 17.4 s a shard; on a 16 GB card that is 10.95 GB server + 1.7 GB miner = 12.7 GB plus the display, under the 15.0 GB line | off on the shipped server, with the line; on (mines and proves, 17 s a v1 shard, 4.3x the alone time) once the patched server ships (the packaging row below) |
| 12 GB (RTX 3060, 4070) | the shipped server: nothing (the floor is 13.9 GB, and the server refuses the card outright); the patched server v3 b37defef at threshold 2^26 (`SP1_GPU_ELEMENT_THRESHOLD=67108864`, the 12 GB profile): **the v1 shard 10,291 MiB and 5.7 s, an empty shard 9,971 MiB and 3.3 s**, the card's 2,089 MiB idle inside the peak and the server's own working set about 8.2 GB (6,535 MiB after Setup), so a 12 GB card proves alone with about 3 GB over it (measured by the prover-floor agent on the 5090's allocation, `floor-sweep-3`; the on-order RTX 3060 run is pending) | measured beside the miner (`floor-sweep-4`): at 2^26 the v1 shard 12,066 MiB total with the miner's 3,833 MiB inside, the server's own 8,233 MiB, 24.4 s; the empty shard 11,586 MiB, 13.0 s; 2^25 gains nothing (12,066 MiB, 43.5 s). On a 12 GB card that is 8.2 GB server + 1.7 GB miner = 9.9 GB before the display, over the 9.0 GB line the project lead set, so mine-and-prove on 12 GB is NOT claimed | off on the shipped server; "proves alone" on the patched one once it ships (the packaging row below); mine-and-prove stays off on 12 GB (9.9 GB plus the display, over the 9.0 GB line). The public gate stays "12 GB proves; 16 GB mines and proves", both on the patched server, both pending a run on the card itself. the project lead's "make sure we can prove on 12 GB cards" is answered on the 5090's allocation and OPEN on the card itself: the prover-floor agent (branch prover-floor, 5 October night) read SP1 v6.8.1's GPU server source (`sp1-gpu/crates/prover_components/src/builder.rs` lines 35 to 39): it reads the card's memory, adds 4 and panics under 24 ("Unsupported GPU memory ... must be at least 24GB"), and builds its core (ELEMENT_THRESHOLD 2^28 + 2^27 elements + 2^21), recursion (2^27), shrink (2^25) and wrap (85 M element) provers at Setup whatever the mode, which is the 13.9 GB floor; no knob reaches them, so the fix is a server rebuilt from source on PC 2 (WSL2, nvcc 12.8, CUDA_ARCHS=120) with those sizes cut, measured on the same fixtures and recipe as the curve above (D2 carries the curve) |
| under 12 GB | nothing | nothing | off, mine only |
| AMD-only and Apple machines | nothing on the GPU: no zkVM proves on an AMD GPU today (`docs/analysis/amd-proving.md`, branch amd-prove); the CPU prover is about 5 minutes a shard at a 30 GB RSS whatever the shard size (PC 1, bench-log "the SP1 CPU prover on PC 1") | | off, "mines and does not prove"; the only non-NVIDIA path with a shipped backend is RISC Zero's Metal prover behind the `ProofSystem` seam (a second guest and pinned id, a verifier for both formats, no shared aggregation): an open item, not 0.3.11 |
The three profile numbers the coordinator asked for, as measured: under 9.0 GB does not exist on this build (floor 13.9); under 15.0 GB mine-and-prove does not exist for any full shard (the v1 shard alone is 20.4); the full profile is the 32 GB card. The fleet table's "proving-only" rows therefore read 24 GB cards at the v1 budget and 32 GB cards at the prototype budget. Shards per block at the v1 budget: 1 on tonight's empty chain, 2 to 4 on blocks with transactions (`B_p` 120,000 = 4 x `S_p`); the aggregation count is one per block whatever the shard count (the chained recursion), so the aggregation-cost agent's target is per block.
The aggregation-cost agent's first rows (branch agg-cost, 5 October 2026 night, the same four live blocks on PC 2): with the miners paused an empty shard proves in 1.9 to 2.2 s and an aggregation in 1.7 to 2.2 s with the card 15.8% busy; mining, 7.4 to 7.8 s and 7.9 to 9.8 s at 93.9% busy, so the miner's kernels take the card and the prover runs 3.6x (shards) to 4.5x (aggregations) slower beside them; its batch-size curve is still open. That puts a proving-only card at about 4 s per empty block (one shard and one aggregation), 4 cards for an empty-block chain at 1 block/s, against 18 mining cards.
The re-plans of block 344 at 2.25 M and 4.5 M pgas peak at 28.3 to 28.4 GB alone (the server's buffers step up between 4.7 M and 20 M cycles and are flat to 60 M), so no shard size between the v1 budget and the prototype one changes a tier; with the miner the adopted shard proves 3.1x slower (13.2 s against 4.2 s) and the chained aggregation 9.7 s against 2.5 s: a mining 24 GB card delivers one adopted-size shard plus one aggregation in about 23 s, inside T by 25x.
## Segment-aligned proving (6 October 2026, the project lead: "find a way to solve this")
**The fault was the work order, not the rules.** The shipped loop took the newest open shard each pass (`prover::choose`), so one prover scattered one block in about 45 across the segment grid and no segment ever had all its blocks proven: pending 56, proven 0, unproven 20 at 06:28Z. No consensus parameter moves.
**The change (app branch, commit ce8f34a, app and host):**
| Part | What it does | Where |
|---|---|---|
| Whole-segment claiming | the work list (lookback 600, the record window) grouped into whole untouched segments: every block present, every shard open (past its 10-DAA exclusive window), unpaid, not in our pool | `app/igneum-app/src/segments.rs` `whole_segments` |
| The choice | candidates inside their deadline by a margin (240 DAA, or 1.5x the last segment's wall time), ranked by FNV-1a of (first block, this prover's key hash): deterministic per prover, different between provers, so several provers spread over the candidates with no coordinator; an attempted segment is not retried | `segments::candidates`, `rank`, `need_daa` |
| The statement check | for the best three: executed and pending; fresh only when the previous segment is not paid and no verified record of it waits in the pool (the chain rule would refuse a fresh record once that one pays); chained (`--prev`) when the previous is paid and its proof is in this node's pool | `prover.rs` `pick_segment` |
| The work | one export to the segment's last block, one fixture per block, one host run `--mode chain --save-shards [--prev]` that proves every shard and aggregates the segment in one process (one key setup), then every shard record signed and submitted (the shard payouts, 90% of the credit) and the segment record signed and submitted (the aggregator share, 10%) | `prover.rs` `prove_segment` |
| The fallback | when no whole segment qualifies, the per-block path as shipped | `prover::choose` |
| The host | `--mode chain` takes `--prev <aggregated.bin>` (chain_len continues; a previous proof that is not the parent block's is refused by number and parent hash) and with `--save-shards` writes per-shard records (statement, proof sha256, file, time) into the results | `proving/igneum-prove/host/src/main.rs` `run_chain` |
| The tile | "Segments: N proven whole, M paid (x IGN to the aggregator), the last in T s" and the segment path's last line; the state carries `segments_submitted`, `segments_paid`, `segment_paid_wei`, `segment_last_s` | `ui/app.js`, `state.rs` |
Tests: the grid, the grouping (missing block, paid shard, our shard in the pool, exclusive shard), the margin and the attempted set, the per-key order (deterministic, different between two keys), the margin from the last time: 6 unit tests in `segments.rs`; 120 app tests, 8 core and 9 host tests pass. The host flags were run on the Mac's CPU first (bench-log, 07:12Z to 07:17Z): per-shard records written for a chain of 2, then a chain of 1 continued from it (`base_chain_len` 2, final `chain_len` 3).
**Item 2, "own pool only", answered from the source:** a relayed proof record carries its proof bytes (`protocol/flows/src/v10/proving.rs`: `IgneumProofRecordMessage { record, proof }`, 8 MB bound, handed to the pool with `local = false`), and so does a segment record (message 75). So "this node's pool" is every record relayed to it, and any aggregator can already fold any 8 proven blocks it has received; no node or consensus change is needed for that. What does limit carriage: a block template carries only entries the node's own verifier marked verified (`template_segment_section`, `template_section`), so a node with the verifier `Off` (node 1) never carries a record and a node with `Trust` carries unverified ones; PC 2's node runs the host and verifies. With one prover the carrier is PC 2's own next block.
**What one 5090 completes (arithmetic from the 5 October rows, the measurement below replaces it):** a chain of 8 empty blocks took 135.6 s cold beside the miner; one export and one key setup per segment instead of eight; so about one segment every 150 to 200 s, 9 to 12 segments an hour out of 450 (2 to 3%), against none. The aggregator share of a segment is 8 x 0.088 IGN = 0.70 IGN plus the 8 shards' 90% share; the forfeited share of the other 97% stays in the escrow until the fleet grows (47 mining cards or 6 dedicated provers for 100% at 1 block/s).
**PC 2 measurement (job `segments-pc2-pv1c`, 07:52Z to 08:24Z, `tools/proving-v1/pc2-segments.ps1`; bench-log "the segment-aligned prover beside the miner"):**
| Figure | Value |
|---|---|
| Whole segments proven in 30 min, one 5090 beside its miner | 9, one every 210 s (export 1.5 s, cut 45 s, chain 160 s: 8 shards 63 s, 8 aggregations 80 s) |
| Shard records accepted and paid | 72 of 72, 0.91 to 2.72 IGN a shard (90% of the credit), carried 180 to 226 blocks after the block |
| Segment records accepted | 0 of 9: refused by the chain rule as shipped (below) |
| Miner's cost | 117.86 to 104.90 MH/s, 13.0 MH/s = 11.0% (the shipped prover: 5.0 MH/s, 4.0%, for 2.8x fewer shards) |
| GPU memory peak | 16.5 to 17.6 GB with the miner resident; 24 GB tier unchanged |
**The second fault, found by the measurement: the chain rule as shipped.** `check_segment_record` accepts a fresh record (chain_len = N) only when the previous segment is UNPROVEN at the carrier. A segment's own deadline is its last block's DAA plus 600, the previous segment's deadline is 8 DAA earlier, so a fresh record is valid for 8 DAA (8 s on devnet) and must be proven before and carried inside them. The refusal on the live node, 9 times: "segment 114470..114477 does not chain to segment 114462..114469 (chain_len 8), which is pending until DAA 169681". The fast-time harness passed on 5 October because its chain continued from proven segments (case 2) and its fresh case ran exactly in that window (case 3, 4 DAA at N = 4). No prover-side move escapes it: the record must be proven, submitted and carried inside the window, which the 210-s proof cannot meet.
**The fix (fork branch 0f0dda95, behind a switch, never by default):** `proving_v1_fresh_rule_daa`; from it a fresh record is valid whenever the previous segment is not proven (pending or unproven) at the carrier; after a proven one a record must still chain. Two records that do not chain each attest their own blocks against the native statement, so nothing is lost but the longer proof chain, which restarts. In the consensus digest only once set (the v1 pattern), so a 0.3.12 node on the live devnet keeps the 0.3.11 digest until the override file sets it; `igneum_getProvingStatus.v1.freshRuleDaa/freshRuleActive` and `igneum_getSegmentStatement.freshAdmissible` report it. Tests: the digest moves once set; fresh after a pending segment passes from the switch, is refused before it, never after a proven one; the harness `--fresh-rule 0` inverts case 3. This is a rule change in the execution layer, not a parameter tuning, and the code proved it unavoidable: the project lead's "no consensus parameter change unless the code proves it is unavoidable" is met by the 8-DAA window above and the nine refusals. The 0.3.12 coordinator sets the height (the same tip + 14,400 rule) in the override object with the release.
**Until the switch:** the app (272b025) holds a refused segment record and offers it again every pass until the segment's deadline, so on the shipped rule it lands only if a carrier falls inside the 8-DAA window, and from the switch it lands on the first retry; the shard records (90% of the credit) land either way, 8 per segment.
**Per tier, with this change and the switch:**
| Card | What it does | Per 30 min, empty blocks (measured on the 5090, approximate elsewhere) |
|---|---|---|
| 32 GB mining and proving (5090) | 9 whole segments, 72 shards paid, 9 aggregator shares once the switch is set | miner 11.0% down; 72 x 0.9 IGN = 65 IGN of shard payouts measured, plus 9 x 0.70 IGN aggregator share from the switch |
| 24 GB mining and proving | the same path at the 16.5 to 17.6 GB peak measured; the fee-switch shard not yet measured on a 24 GB card | approximate: the 5090's numbers |
| 16 GB | mines and proves on the patched server only (prover-floor rows); segment path untested there | pending the prover-floor agent's build |
| 12 GB prove-only | proves alone on the patched server (10.3 GB); a dedicated prover takes 4 s a block alone (agg-cost rows), so about one segment every 40 s | approximate: 45 segments per 30 min, 6 such cards for 100% |
| A rig (several cards) | one prover loop per machine today; the segment path claims one segment at a time on the aggregation card | the per-card loop is the next item |
## v1 live on devnet (6 October 2026, C47)
v1 active at 154,800 (crossed at DAA 154,814, 03:51:42Z); first segment record: none, because on a one-prover devnet no segment can be proven. The app's aggregator (`aggregate_once`, 0.3.11) needs a shard proof of every shard of every block of the segment in its node's pool, and PC 2 alone proves 13 shards per 10 minutes of about 600 blocks (2.2% coverage), so a run of 8 consecutive proven blocks never occurs: node 1 at 04:16Z reads segmentsInWindow pending 55, proven 0, unproven 20, paidSegments 0, and PC 2's app log (run win-1ccfe586-20261005-235130, 04:11Z to 04:16Z) reads every 42 s "aggregator: segment N..N+7: waiting for shard proofs N/0 ... N+7/0 in this node's pool", all 8 missing, each segment then past its 600-DAA deadline unproven. No fault in the node, the app or the record path; the fast-time harness passed because its shards ran at 90% coverage. Meanwhile the aggregator share (a tenth of every block's pool credit) stays in the escrow; shard payouts continue; miners and block watchers see nothing. What ends it: coverage at 8 consecutive blocks, 47 mining 5090-class cards with the shard loop as shipped in 0.3.11 (13 shards per 10 minutes a card), 18 mining cards through the chain mode, or 6 (approximate) proving-only cards, at 1 block/s on empty blocks (the fleet table above), or the segment length lowered on a small devnet (`proving_v1_segment_blocks`, a consensus param, so a digest change). No PC 2 job and no 0.3.12 item follow from this; the open item is the fleet, not the code.
## The empty `/api/state` reply (6 October 2026)
The aggregation-cost agent's jobs read the two bytes `{}` from `/api/state` on PC 2 at 22:22Z, 22:41Z and 00:18Z (0.3.10 and 0.3.11); the 21:01Z reply was full. Cause, from the app source and node 1's RPC: `ProvingState.paid_wei` is a `u128`, and serde_json's `to_value` refuses a u128 over u64::MAX (18,446,744,073,709,551,615 wei, 18.45 IGN); `state_json()` turned that refusal into `json!({})` with no log line. A paid shard is 1.23 IGN on average (node 1, `igneum_getProvingStatus`: 814.64 IGN over 663 shards at 00:3xZ), so the fifteenth paid shard after an app start empties the reply. PC 2's prover was blind to the root-owned socket from 20:00:56Z to 22:01Z (paid_wei stayed 0, hence the full reply at 21:01Z), proved from 22:02:13Z, and crossed 18.45 IGN inside its first 15 paid shards, before 22:22Z. Every app restart resets the counter, so the reply comes back for about 15 shards and goes again.
What it means: the dashboard on a proving machine shows nothing within about 12 minutes of its prover's first payout; every PC playbook that reads a card from `/api/state` fails the same way (the agent's job 5 reads settings.json instead). Mining, proving and payouts are untouched; it is the status page only.
Fix on the app branch: `paid_wei` serialises as a decimal string (the dashboard already reads it with `Number()`), `state_json` logs `[error] state_json: ...` once instead of answering `{}`, and the reply on any future serialisation error carries `error` and `version` rather than nothing; unit test `a_paid_total_over_u64_max_still_serialises_the_whole_state`. Not in 0.3.11 (that tree closed at 22c2363, master 630da6b, published); 6714a45 heads 0.3.12, the morning's first cut, app only, before PC 1's relaunch (coordinator, counter-asic-2-rollout.md 8a); until then the workaround is settings.json for the card keys.
## Aggregation cost (5 October, night)
the project lead, 5 October 2026: "fix everything else in the numbers tonight". Branch `agg-cost`; every number in `docs/bench-log.md`, "aggregation cost on the RTX 5090", with its job id. The proof statement and the pinned guests are unchanged: every existing fixture proof still verifies (`verify-segment` 0.027 s on the Mac, 0.036 to 0.041 s on PC 2).
| What | Before (5 October evening, `chain-pc2-pv1c`) | After (5 October night) |
|---|---|---|
| Chained aggregation, the card mining | 9.6 to 9.7 s a block | 9.6 to 9.8 s a block, the same (job `agg-cost-pc2-1`, phase A); the miner's presence is the whole cost: 2.1 to 2.2 s a block with the card to itself, 1.7 s unchained |
| Shard proof (empty shard), the card mining | 7.3 to 7.7 s | 7.4 to 7.8 s; 1.9 to 2.2 s with the card to itself |
| The miner's slowdown of the prover | 3 to 4x (against 4 October) | measured on the same fixtures 2 min apart: shards 3.6x, chained aggregation 4.5x, a block 4.2x |
| Where the time goes | not profiled | the host's share 0.000 s (the prove call is everything); the GPU server prints no timings; the second deferred proof (the chain rule) costs 0.4 to 0.5 s alone and 1.7 to 1.9 s under the miner; the prover alone keeps the card busy 15.8% of the time, the miner 93.9% |
| SP1 knobs (`SP1_WORKER_VERIFY_INTERMEDIATES=false`) | not tried | no gain: 7.8 s against 8.1 s over four aggregations, inside the spread; the shape knobs would change the recursion keys the pinned verifier accepts |
| Batch fold (K blocks in one aggregator call) | not estimated | estimate from the measured step costs: 0.9 s a block alone and 3.8 s mining at K = 4, 0.7 and 2.8 s at K = 8 (0.25 s per further deferred proof alone, 1.8 s mining); a tree fold gains nothing. A new pinned guest and program id either way, so not tonight |
| Two prover processes on one card | not tried | closed on SP1 6.8.1: both share one GPU server socket, run slower together (6.9 s a block against 4.1) and the second dies with the first (`early eof`) |
| The miner's kernel length (`--batch-log2` of the CUDA worker, 2^B nonces a launch; job `agg-cost-pc2-6`, the 5090 alone with the job's own miner) | not tried | 2^22 (the default) and 2^20: 18.1 and 18.0 s a block, 10.0 to 10.4 s a chained aggregation, 104 MH/s; 2^18: 15.6 s, 8.8 s, 99 MH/s (minus 4%); 2^16: 11.1 s, 6.1 to 6.2 s, 84 MH/s (minus 19%), reproduced |
| The GPU time-slice policy (`nvidia-smi compute-policy --set-timeslice`) | not tried | "Not Supported" on PC 2 (driver 13.3, Windows): closed |
| The chosen combination | the defaults | the defaults stay: batch-log2 22 and SP1's default knobs. The one knob that moves the prover (2^16) costs a fifth of the hash rate all the time for a prover that is busy a few seconds a minute on the devnet; it is the project lead's trade, not a default (below) |
Reading. The per-block aggregation is 2.1 s and a block 4.1 s on a 5090 that only proves, 9.7 and 17.5 s on one that also mines; no knob, fold or stream on tonight's SP1 changes the first pair, and only the miner's kernel length changes the second, at 1 MH/s per 0.37 s of block time. So "under 3 s a block" and "under 1.5x" are met on a card that is not mining and are not reachable on one that is. What that means per tier: a 5090 that mines and proves delivers a proven empty block every 17.5 s (6 cards for 1 block/s), the same card proving only every 4.1 s (2 cards, plus the shard work of full blocks: the fleet table above), and a batch fold of the aggregator (a new pinned guest) would bring the proving-only card to about 2.7 s a block and the mining one to about 12 s. What is being done: the app and host defaults are left as measured; the plan's open decision for the project lead is whether a card that holds a shard assignment should drop to 2^16 for the proof's minute (1.6x faster proof, 19% of its hash rate for that minute) or whether proving-only cards carry the aggregation (the clean 2.1 s), and the batch fold goes on the next pin's list. The state class found on the way (`/api/state` answering `{}` once `paid_wei` passes u64::MAX, fixed on the app branch at 6714a45) is in the bench log with the rest.

126
docs/plans/read-width.md Normal file
View file

@ -0,0 +1,126 @@
# Read width of the lottery hash: 4, 16 and 64-byte loads, a per-load mix, and a written scratch (gate 1 experiment)
5 October 2026. Branch `readwidth` (worktree `../igneum-wt-readwidth`), commits 019b014 and b970dda plus the measurement commit. Nothing here changes consensus, the live generator, the pinned vectors or a shipped binary: every class sits behind `--class` in `igneum-pow` and the default class is generator version 2 byte for byte (`igneum-pow/tests/packs.rs` still compares the four pinned packs against the emitters). Numbers and a recommendation; the decision is the project lead's.
## 1. The question
The bench-log entry "the 9070 XT on the eGPU" (5 October 2026) found the hash bound by dependent random 4-byte reads over the 1 GiB dataset, 128 per hash: the RX 9070 XT finishes 2.4 to 2.7 G such reads a second (18 MH/s), the RTX 5090 16 to 18 G (127 MH/s), the M5 Max 3.45 G (23 to 28 MH/s). AMD fetches a 64-byte line per 4-byte read, so 94 percent of its memory traffic is unused; NVIDIA fetches a 32-byte sector and its 96 MB L2 catches a share. the project lead's question: would wider reads keep the chip-resistance property (latency-bound, random access) while closing the vendor gap? Two additions from the coordinator: a per-load width drawn from an era-fixed mix so no chip is built for one width, and a written per-warp scratch so part of the memory work cannot be mirrored into read-only SRAM.
## 2. What was built (all behind the flag)
| Class (`--class`) | Loads per hash | What a load does | Dataset bytes per hash |
|---|---|---|---|
| `v2` (= `w4`, the lottery hash) | 128 | `dst ^= dataset[src & MASK]`, one 4-byte word | 512 |
| `w16` | 128 | the 16-byte-aligned group of 4 words at `src & MASK`, every word folded into `dst` | 2,048 |
| `w64` | 128 | the 64-byte-aligned item (16 words), every word folded | 8,192 |
| `w64x4` | 32 (4 load slots) | as `w64`; the same bytes per hash as 512 loads of 4 bytes | 2,048 |
| `mix50-35-15` | 128 | per load, width 4, 16 or 64 bytes drawn from the program stream with probabilities 50/35/15 | 1,664 to 3,680 over the six programs measured (expected 2,202) |
| `mix25-50-25` | 128 | the same with 25/50/25 | 2,240 to 5,024 (expected 3,200) |
| `scr<k>k<kb>` | 128 memory operations | `k` of the 16 slots are scratch read-modify-writes into a `kb` KiB per-warp scratch (16-byte slots, lane-major); the other `16 - k` are 4-byte loads | 4 x (16 - k) x 8 reads plus 16 B read and 16 B written per scratch op |
The fold. A load of W words reads the W-word-aligned address `b = (src AND MASK) AND NOT (W - 1)` and sets `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) XOR w[j]; dst = x` (`verify::fold_words`, mirrored in the three kernel dialects). The rotate-multiply between the words makes the fold state-dependent: two different lines give two different maps of `dst` (the multiply by an odd constant is not xor-linear), so no function of the line alone can stand in for it, and a dataset of pre-folded lines cannot replace the dataset. The dependent chain is unchanged: the next load's address comes from a register that the fold wrote. Width 1 is the lottery hash's xor of one word, so `w4` is the pinned program `bcc1248b10cc90f2` exactly.
The per-load draw. Every class other than `v2` takes one extra draw per instruction (`below(100)`, the width roll, consumed on every slot so the stream stays uniform) after the nine draws of spec 01 section 1.4.3; on a load slot the width is the first entry of the mix whose cumulative weight exceeds the roll. The program id of a class is `FNV-1a-64("igneum-program-rw/" || 2 || seed words || attempt || mix[3] || slots [|| "scratch/" k kb])`, so no class program can pass for a version 2 program. The acceptance rule of 1.4.6 runs unchanged on the aligned addresses (lane-constant sites and the distinct-address bound, scaled to the dataset loads per hash).
The scratch (variant 5, measurement only). The kernel runs N persistent warps (one per block or work-group of 32); warp `w` owns scratch `w` and runs units `w, w + N, ...` of the launch. A scratch op reads the lane's 16-byte slot `src AND (slots - 1)`: three data words behind a per-unit tag; a slot whose tag is not this unit's reads as its fill `splitmix32(((base + lane) XOR seed[j]) + slot x 0x9e3779b1 + (j + 1) x 0x85ebca77)`, the three words are folded into `dst` as above, and the slot is rewritten `(tag, x XOR w1, rotl(x, 7) XOR w2, x + w0)`. The CPU verifier holds the touched slots of one unit (at most 32 x k x 8) and nothing else. The scratch is per lane (a 32 KiB warp scratch is 64 slots per lane, 128 KiB is 256), so two lanes never race on a slot and the result is a function of (program, day, unit) alone; the GPU's tags make the lazy fill exact as long as a tag is not reused within the arena's history (the salt advances per unit; it wraps after 2^32 units, a measurement caveat, not a design).
## 3. Method
| Step | Command (every figure in the bench log carries its command) |
|---|---|
| Packs | `igneum-pow export --seed <s> --class <c> --out proto-cuda/packs-readwidth/<name>` (23 packs; the six mix seeds per mix are `igneum-readwidth/A/<k>` and `/B/<k>` with attempt 0 accepted, `A/4` skipped: rejected at attempt 0) |
| CPU verifier | `igneum-pow bench --seed igneum-genesis --class <c> --warps 50` (M5 Max, one core; load average 4 to 9 from other agents' builds during the run) |
| Metal | `proto-metal/packbench --pack <dir> --batches 5 --batch-log2 24 --group 256 [--warps N]` (new harness: runs the pack's own text; vectors, cache FNV, dataset words, 2^24 fingerprint, MH/s by GPU time) under the measure lock |
| Apple OpenCL | `proto-opencl/igneum-bench-cl-rw --bench-pack --pack <dir> --batches 5 --batch-log2 24` and `--memprobe` (Apple's OpenCL, a correctness check and an approximate rate) |
| Emulators | `proto-cuda/emu/emu.sh ../packs-readwidth/<p> --batch-log2 13 --batches 1 --block-warps 2` (the CUDA text as C++); `proto-opencl/emu/emu.sh ../packs-readwidth/<p> 0 32 --sg 32` and `1 64 --sg 64` (the OpenCL text, wave32 and wave64) |
| RTX 5090 | job `run-readwidth-5090-20261005` on PC 2 (`relay/playbooks/readwidth-5090.ps1`): the NVIDIA card switched off in the app through `POST app.url/api/cards` and restored after; `igneum-worker-cuda --memprobe`, then `--bench --pack <dir> --batches 5 --batch-log2 24` per pack (NVRTC, the pack's own text, vectors through the bound kernel, 2^24 fingerprint) |
| RX 9070 XT | job `run-readwidth-9070-20261005` on PC 1 (`relay/playbooks/readwidth-9070.ps1`): only the gfx1201 card switched off; `igneum-worker-opencl --device D --memprobe`, then `--bench-pack --pack <dir> --batches 5 --batch-log2 24` per pack |
Latency-bound share = measured MH/s x loads per hash / the card's dependent-read ceiling for that width from its own probe at 1024 MiB (for a mix, the harmonic combination of the widths' ceilings weighted by the program's width counts). A share near 1 means the hash runs at the card's random-access limit, the property the design wants; a share well under 1 means something else bounds it (bandwidth, ALU, occupancy).
## 4. Results (full tables with commands in `docs/bench-log.md`, "read width of the lottery hash")
Probe ceilings at 1024 MiB (G dependent reads/s): RTX 5090 4 B 17.5, 16 B 18.0, 64 B 9.1 (584 GB/s), stream 1,579 GB/s; RX 9070 XT 4 B 2.42, 16 B 2.43, 64 B 2.47 (158 GB/s), stream 636; M5 Max (Apple OpenCL, approximate) 3.50 / 3.51 / 3.51, stream 522.
| Class | dataset B/hash | RTX 5090 MH/s (latency-bound share) | RX 9070 XT (share) | M5 Max Metal (share) | 5090 / 9070 | DRAM bytes moved per hash, NVIDIA 32 B sector / AMD 64 B line | CPU verify ms per unit |
|---|---|---|---|---|---|---|---|
| v2 = w4 (today) | 512 | 136.1 (0.96) | 18.15 (0.87) | 27.74 (1.01) | 7.5x | 4,096 / 8,192 | 0.604 |
| w16 | 2,048 | 139.8 (0.90) | 17.90 (0.84) | 28.26 (1.03) | 7.8x | 4,096 / 8,192 | 0.610 |
| w64 | 8,192 | 71.9 (0.58) | 17.59 (0.78) | 28.27 (1.03) | 4.1x | 8,192 / 8,192 | 0.630 |
| w64x4 (32 loads) | 2,048 | 275.3 (0.56) | 75.19 (0.84) | 109.7 (1.00) | 3.7x | 2,048 / 2,048 | 0.160 |
| mix50-35-15 (6 programs, min / median / max) | 1,664 to 3,680 | 99.5 / 114.2 / 121.0, spread 18.8% | 17.45 / 18.76 / 18.83, 7.4% | 25.36 / 27.26 / 28.43, 11.3% | 6.1x | 5,939 / 8,192 expected | 0.620 |
| mix25-50-25 (6 programs) | 2,240 to 5,024 | 95.9 / 107.3 / 119.8, 22.3% | 17.84 / 18.45 / 18.85, 5.5% | 23.21 / 24.68 / 25.21, 8.1% | 5.8x | 7,168 / 8,192 expected | 0.614 |
### 4.1 Per watt and per pound (consequences review C11)
The runs carried no power sampling; the watts are the telemetry entry's (`docs/bench-log.md`, opencl-rdna4-telemetry, 5 October 2026: the RTX 5090 at 307.6 W under its 450 W cap for 122.3 MH/s, the RX 9070 XT at 199 W of its 304 W rating for about 17.8 MH/s, both on v2 with the shader clock at its top and the die waiting on memory), held constant across classes because every class is memory-bound on both cards (approximate: a class that moves more bytes per hash draws somewhat more at the memory controller, unmeasured). The Mac's GPU power is not measurable without root (`powermetrics`) and is taken as about 50 W (approximate, from memory). Prices are UK list, approximate, from memory.
| Class | RTX 5090 MH/W (at 307.6 W) | RX 9070 XT MH/W (at 199 W) | 5090 / 9070 per watt | M5 Max MH/W (at about 50 W GPU, approximate) | 5090 MH per pound (at about 1,900, approximate) | 9070 XT MH per pound (at about 570, approximate) |
|---|---|---|---|---|---|---|
| v2 (w4) | 0.442 | 0.091 | 4.9x | 0.55 | 0.072 | 0.032 |
| w16 | 0.454 | 0.090 | 5.1x | 0.57 | 0.074 | 0.031 |
| w64 | 0.234 | 0.088 | 2.6x | 0.57 | 0.038 | 0.031 |
| w64x4 | 0.895 | 0.378 | 2.4x | 2.19 | 0.145 | 0.132 |
| mix50-35-15 (median) | 0.371 | 0.094 | 3.9x | 0.55 | 0.060 | 0.033 |
| mix25-50-25 (median) | 0.349 | 0.093 | 3.8x | 0.49 | 0.056 | 0.032 |
| scr8k32 | 0.397 | 0.071 | 5.6x | 0.98 | 0.064 | 0.025 |
| scr2k32 | 0.372 | 0.074 | 5.1x | 0.52 | 0.060 | 0.026 |
Reading: whatever width is chosen, an AMD home miner keeps about a seventh of a 5090's rate and pays about 4.5x the electricity per hash, because every width costs the 9070 XT the same 2.4 G line fetches a second; per pound of card the 5090 is 2.2x the 9070 XT at v2 and w16 (0.072 against 0.032 MH/s per pound) and 4.9x per watt; only w64x4 narrows the per-pound gap (0.145 against 0.132), and that class fails the width rule. The consequence for the decision (D6, the project lead's): AMD's line width is not a read-width question at all; it is the card's random-access rate, and the levers that act on it (the 64 MB Infinity Cache against the dataset size, the memory path) are v3-or-3.0 questions outside this experiment.
Scratch, variant 5 (N persistent warps; GPU cost against the persistent control scr0k32; working set = 1 GiB + 256 MiB + 128 MiB output + N x size):
| Class | RMW share | dataset B/hash | scratch B/hash read + written | RTX 5090 MH/s, 2,048 warps of 4,080 resident (vs control, share) | M5 Max Metal (vs control) | RX 9070 XT, 4,096 warps | working set 5090 / 9070 / Mac |
|---|---|---|---|---|---|---|---|
| scr0k32 | 0 | 512 | 0 | 139.1 (control, 0.98) | 28.25 (control) | 17.88 (control, 0.86) | 1.4 GiB / 1.5 GiB / 1.5 GiB |
| scr2k32 | 12.5% | 448 | 256 + 256 | 114.4 (-18%, 0.80) | 26.14 (-7%) | 14.65 (-18%) | same |
| scr4k32 | 25% | 384 | 512 + 512 | 109.8 (-21%, 0.76) | 31.74 (+12%) | 14.00 (-22%) | same |
| scr8k32 | 50% | 256 | 1,024 + 1,024 | 122.1 (-12%, 0.82) | 49.08 (+74%) | 14.17 (-21%) | same |
| scr2k128 | 12.5% | 448 | 256 + 256 | 110.1 (-21%, 0.77) | 26.24 (-7%) | 14.07 (-21%) | 1.6 GiB / 1.9 GiB / 1.9 GiB |
| scr4k128 | 25% | 384 | 512 + 512 | 98.0 (-30%, 0.68) | 28.08 (-1%) | 13.14 (-27%) | same |
| scr8k128 | 50% | 256 | 1,024 + 1,024 | 72.8 (-48%, 0.49) | 35.44 (+25%) | 12.03 (-33%) | same |
Resident warps and the cap: the 5090 holds 4,080 warps at one warp per block (24 blocks per SM x 170 SMs; 8,160 at 8 warps per block), so 128 KiB each is 510 MiB and the whole working set 1.9 GiB; a 1 MB scratch would have been 4.0 GiB at this geometry and 10.6 GiB at the 64-warp-per-SM figure, which is why the cap moved the size to the tens of kilobytes. The occupancy query returned 24 blocks per SM before and after the arena allocation: the allocation did not change it. The 9070 XT's OpenCL runtime has no occupancy query; 4,096 persistent warps were launched (64 per compute unit over 64 CUs, approximate) and the arena is 128 MiB at 32 KiB, 512 MiB at 128 KiB. The Mac's residency is not reported; 2,048 to 16,384 warps were swept and the best row kept.
Chip model, re-run with the measured widths (the M16 arithmetic of `docs/analysis/m16-recompute-attacker-2026-10-05.md`; the on-die-cache recompute chip's row per scratch variant is the ca2-soundness branch's, as agreed with the Counter ASIC 2.0 coordinator):
| Class | what a chip with its own DRAM controller gains over the GPU's memory system | what a chip with on-die SRAM gains |
|---|---|---|
| v2 | the GPU fetches 8 to 16x the bytes it uses (AMD 64 B, NVIDIA 32 B per 4 B); a chip fetching 32 B bursts moves 4,096 B per hash, the 5090's figure, so nothing over NVIDIA and 2x over AMD in traffic, none in latency (the chain is 128 dependent DRAM latencies on either) | the recompute attacker of M16: 150,000 integer ops per hash against the 256 MiB cache; 2.4x at equal silicon before a fixed-function factor (unchanged by the width) |
| w16 | the same: 4,096 / 8,192 bytes moved, 2,048 used; traffic efficiency 50 percent on NVIDIA, 25 on AMD | unchanged: the fold uses every byte, so the chip recomputes 128 items per hash exactly as before; the SRAM mirror of the read-only dataset (1 GiB) stays out of reach |
| w64 | every byte moved is used on both vendors (8,192 moved, 8,192 used); the 5090 is bandwidth-bound at 589 GB/s, so a chip with HBM3 class bandwidth (several TB/s, approximate) is bandwidth-advantaged: the Ethash shape | unchanged in op count; but the chain of 128 loads now moves 8 KB, so a chip's advantage shifts from latency to bandwidth per dollar, which is the wrong direction for the design's 2x target |
| w64x4 | 2,048 moved and used; 32 latencies per hash; every card 4x faster; a bandwidth-rich chip gains as above | 32 items per hash: the recompute attacker's op count falls 4x (37,500 per hash), so the M16 gain rises 4x: fails the 2x target by arithmetic |
| mixes | between v2 and w64 per program; the chip cannot be built for one width, but the GPU pays the 64-byte hours (the 5090 loses up to 27 percent in a heavy hour) | as v2 per item; the recompute attacker is indifferent to the width |
| scratch | a chip must provide writable memory for N units in flight: 32 KiB x N at the GPU's geometry (128 MiB at 4,080), against the 256 MiB read-only cache it could mirror into SRAM (54 to 83 mm^2 at a leading node, the coordinator's figure, approximate); but a unit touches at most k x 8 x 32 slots (4 KiB at 50 percent), the fill is a function and the tags are per unit, so a chip need only hold the touched set per unit in flight (the soundness caveat below) | the dataset reads replaced by scratch ops are reads the chip no longer has to serve from the 1 GiB; at 50 percent the recompute attacker computes 64 items instead of 128 |
Soundness (variant 5, measurement only, as instructed; the chip row is the ca2-soundness branch's, a465881: the on-die-cache recompute chip's gain is 2.4x at 0, 12.5, 25 and 50 percent, replaced or added, 32 or 128 KB, so the scratch does not move it): the per-unit scratch starts from a fill that any implementation can compute, and a unit writes at most `k x 8` slots per lane; an implementation that keeps only the touched slots of each unit in flight (the CPU verifier does exactly this) needs 16 B x touched slots, not the nominal arena, so the "real memory a chip must provide" is bounded by units in flight x touched slots, not by N x 32 KiB. The variant forces memory that is written, which SRAM can hold as well as DRAM; it does not force memory that is large. A written region that outlives the unit (state carried across units) would, and the CPU verifier could not replay it. This is the finding, not a recommendation.
## 5. Recommendation (the decision is the project lead's)
the project lead's rules, as passed by the coordinator: width = the widest read that keeps every card latency-bound (achieved within 90 percent of the probe ceiling at that width) with margin on the 5090 (bytes per hash x rate under a third of the 1,579 GB/s stream); the mix is in only if the six-program spread is under 5 percent per card; the scratch share is the smallest at which the chip model's gain falls under 1.5x at the lowest GPU cost within the 6 GB cap.
| Variant | Verdict under the rules | Numbers |
|---|---|---|
| w16 (16-byte loads, 128 per hash) | PASSES the rules: shares 0.90 / 0.84 / 1.03 (the 9070 XT's 0.84 equals its v2 share of 0.87 within noise: the card is at its ceiling in both), 286 GB/s on the 5090 = 18 percent of the stream. It does NOT close the vendor gap (7.8x against 7.5x), because the memory systems already move a sector or a line per load; it changes what the fold consumes, nothing the DRAM does | the only width row that passes; a no-cost change in rate (+2.7 percent 5090, -1.4 percent 9070 XT, +1.9 percent M5 Max) |
| w64 | FAILS: 5090 share 0.58, 37 percent of the stream; closes the gap to 4.1x only by making the 5090 bandwidth-bound | |
| w64x4 | FAILS: shares 0.56 / 0.84 / 1.00, the recompute gain rises 4x | |
| mix 50/35/15 and 25/50/25 | OUT: spreads 18.8 and 22.3 percent on the 5090, 7.4 and 5.5 on the 9070 XT, 11.3 and 8.1 on the M5 Max, all over 5 percent; a chip is not built for a width anyway (see the model: the width does not change the recompute attacker) | |
| scratch | OUT: every share costs the 5090 12 to 48 percent and the 9070 XT 18 to 33 percent, and raises the M5 Max's rate (the arena is cached there); the soundness branch's chip row (ca2-soundness a465881, the on-die-cache recompute chip) stays at 2.4x at every share, 32 or 128 KB, because the verifier resets the scratch per unit and the live state is the hash's own read-modify-writes, which a chip keeps in 80 to 320 B per lane; under the rule the share is 0. The rows stay as the measurement that decided it | |
Recommendation: keep 128 loads per hash and 4 bytes per load (v2) for the devnet; if a width change is wanted for the fold's sake (every byte of the sector consumed, which removes the "94 percent waste" statement from the AMD entry without changing what the card does), w16 is the one that passes every rule and costs nothing measurable, and it is the only width worth a vector re-cut. The AMD gap is a random-access gap (2.4 G against 17.5 G dependent reads per second at 1 GiB on the cards we own), and no read width closes it without turning the 5090 bandwidth-bound; the levers that act on the gap are the ones outside this experiment (the AMD card's memory path, and the dataset size against the 5090's 96 MB L2 share, which the probe's 64 MiB rows show at 9 G reads/s against 2.4 at 1 GiB). The per-load mix is out on stability; the scratch is out on GPU cost and on the soundness caveat.
## 6. What w16 would change if adopted (not done; the decision is the project lead's)
| Where | Change |
|---|---|
| `docs/spec/01-lottery-hash.md` 1.4.1 | `load`: `dst = fold(dst, dataset[b .. b + 4))`, `b = (src AND MASK) AND NOT 3`, with the fold written out; 1.4.3 unchanged (no width draw for a fixed width); 1.4.6 unchanged (the aligned address is the address the rule sees) |
| 1.5 | "A load reads one 4-byte word" becomes 16 bytes aligned; the single-form text-search rule of 1.14 item 2 becomes the wide form; the item size (64 B) and `dataset[w] = item(w >> 4)[w AND 15]` unchanged |
| 1.11 | unchanged in count (4,096 items per unit; the verifier derives the same items) |
| 1.15, 1.17 | every vector re-cut (new program ids: the class enters the id or the generator version steps to 3); the four pinned packs replaced; the conformance fuzz re-run on Metal, CUDA and OpenCL (this branch's 23 packs and the three emulators are the template) |
| Litepaper, Mining ("random reads over a multi-gigabyte dataset") and the vs-RandomX "128 dataset addresses" rows | "128 reads of 16 bytes"; `site/bench.html` sector arithmetic (32 B per 4 B) becomes 32 B per 16 B |
| Workers | no host change: the kernel text carries the loads; `proto-cuda/host.cu`'s static mask check (`TESTS.md` section 5) learns the wide form |
| Cost on the 5090 | none measured (+2.7 percent); on the 9070 XT -1.4 percent; verifier +1 percent |
## 7. Files
`igneum-pow/src/{generator,verify,accept,emit,memhard,main}.rs` (the classes, behind `--class`), `proto-cuda/packs-readwidth/` (23 packs), `proto-metal/packbench.swift` (Metal from a pack's files), `proto-opencl/host.c` (`--bench-pack`, `--warps`, the 16-byte probe row, the scratch arguments), `proto-cuda/nvrtc/worker.cpp` (`--bench`, `--memprobe`, the scratch arena), `proto-cuda/nvrtc/packfile.h` (class fields; string-seed packs), `proto-cuda/emu/cuda_runtime.h` and `proto-opencl/emu/{emu_opencl.h,emu_main.cpp}` (vector types, the persistent launch), `relay/playbooks/readwidth-*.ps1` (the PC jobs: the card under test off in the app and restored, never the other card).

View file

@ -0,0 +1,489 @@
# Igneum Miner 0.3.10: the certificate-driven reorg, transaction gossip, the pack loader fix, the six-section miner and the PC-built Windows node, 5 October 2026
Shipped: manifest 0.3.10 published 21:32:40Z, every reachable machine on 0.3.10 by 21:49:41Z, the hand nodes and the seed on 21d4c73c by
21:50:15Z, one digest (1f4b4425...) on every node that restarted; PC 37ba0461 (the owner's laptop) mid-install and Sam's Mac quit at the
time of writing. The cut ran from 16:39Z to 21:5xZ, two hours of it GitHub's hosted-runner outage.
Release engineer, from 16:39 UTC. Worktree `/Users/joshm/Projects/igneum-wt-ship0310`, branch `release-0.3.10` from master
after the 0.3.9 merge. Fork worktree `vendor/igneum-node-0310`, branch `release-0.3.10` from a24ab01a (the 0.3.9 fork tip).
Every Mac build through `/Users/joshm/Projects/igneum/tools/lock/with-lock.sh build` at `nice -n 19` with `-j 4`; every PC
job through the main checkout's `tools/build-job.mjs` and `packaging/ota/publish-jobs.sh` (the signed envelope). Times are
UTC. `release-0.3.9.md` and `release-0.3.8.md` are the template; `release-0.3.6.md` section 10 for the Windows acceptance.
The rollout waited for 0.3.9's (shipped about 17:00Z, fee switch published about 17:50Z): nothing of 0.3.10 moved on the
network before every node's digest was ab8847da...
## 1. What 0.3.10 carries
| Change | Where | Fork commit |
|---|---|---|
| A. The windows-gnu node links the C++ runtime statically (`database/build.rs`, `rocks-probe`), so the PC-built Windows node runs (release-0.3.6 plan section 10) | fork `housekeeping` 4fb32865 | merged 2f88a82f |
| B. `TestConsensus` sets the unbounded PoW cache build queue, so the five node suites pass in parallel on PC 2 | fork `housekeeping` 7003055b | merged 2f88a82f |
| C. The certificate-driven reorg (ledger C4, measured in FUD round 6, 75c6658: a certificate over an off-chain block stayed pending and the heavier side locked alone): a verified certificate over a block off the node's selected chain now locks it there and moves the virtual to the heaviest tip through it (spec 3.5 F1 and F2); the block relay takes a lighter block a pending certificate names. A consensus-behaviour change with NO digest change: it converges when every node is new (the c4 agent's rollout note, section 10) | fork `c4-fix` e18f1e0e (its message records PC 2 job build-20261005-180827: kaspa-consensus 97 passed, kaspa-consensus-core 101 passed, three new tests), main `c4-fix` 68f14b9 (spec 3.5, 3.2, 3.10, 3.11.7, ledger, bench-log, c4.mjs v2 mode; the signer-pipe-check CI script and per-job build-inputs zip names) | fork: merged 21d4c73c; main: 0cbaf3a and the tip merge |
Changelog line for C (the coordinator's words): "A node that comes back from a partition now follows the certified checkpoints, even when its own chain is heavier."
| D. EVM transaction relay between nodes: three p2p messages after Kaspa's transaction inv/request/answer, PROTOCOL_VERSION 13 to 14 (13 peers still connect: a v14 node never sends the new messages to an older peer), the mempool caps and faults, the block-added hold instead of the 4-s hand-out cooldown; the digest is unchanged, so no activation height and no fresh chain | fork `tx-gossip` e242acd0 (its message records PC 2: igneum-exec 15 of 15, kaspa-p2p-flows 33 of 33), main `tx-gossip` 0166e94 (the relay-net harness `tools/txgen/relay-net.mjs`, the design note); acceptance here: `node tools/txgen/relay-net.mjs --rate 2 --duration 120 --wallets 16 --fund 2` on the 0.3.10 Mac node, expected 240 of 240 included | fork: merged 21babaa7; main: (pending the worktree) |
Changelog line for D (the coordinator's words): "Transactions now travel between nodes; a miner on an older node still carries only what was sent to it directly."
| E. GPU hot-plug (coordinator, 17:5xZ): cards re-enumerated every 60 s and on `WM_DEVICECHANGE`, new cards mine by default, Code 43 cards marked unusable, vanished cards dropped without shifting slots, the Ryzen gfx1036 iGPU classified integrated and off by default, a `cards:` line the relay parses (machines API `gpus`/`hotplug`). 14 files: `app/igneum-app/src/{detect,engine,hotplug,main,state}.rs`, `ui/{app.css,app.js,notices.test.mjs}`, `app/windows/host.cpp`, `relay/{api/console.mjs,lib/parse.mjs,test/parse.test.mjs,ui.html}`, `tools/console.mjs`; no proving, packaging or workflow file. Its Windows-only paths (`detect::adapters`, `host.cpp` WM_DEVICECHANGE) were never compiled on the Mac: PC 1's Windows app build and the GitHub runner's host build are their first compile and are read for errors | main `gpu-hotplug` 4517004 (`igneum-wt-hotplug`, on master 22ffc02); its tests on the branch: igneum-app 91 pass, notices, update-card and parse node tests pass | merged after C and D |
| F. Elevated jobs that never started (coordinator, 18:0xZ): an elevated job whose UAC prompt is refused, cancelled or times out fails with exit 251 and the summary "did not start: the administrator prompt was refused, cancelled or timed out"; before, `Start-Process` threw, `$p` stayed null and `exit $p.ExitCode` exited 0, so the PC 1 driver job at 17:58Z read as done. One file, `app/igneum-app/src/jobrun.rs`, plus its unit test | main `elevated-exit` 9816d58 (`igneum-wt-elevated`) | merged after E |
| G. The miner UI redesign (the project lead, through the coordinator at 19:0xZ: into 0.3.10, not 0.3.11): six sections (Mine, Prove, Rewards, Node, Updates, Settings), one switch per card, one line of help per setting, the node's next switch shown with its height; three engine additions with tests (`state.rs` `node.consensus_digest` and `node.consensus_switches`, `engine.rs`, `prover.rs` program ids via `--mode id`), a Mac window minimum of 900 x 600, one `WM_GETMINMAXINFO` case in `app/windows/host.cpp`, `view.test.mjs` in CI | main `miner-ui-2` (`igneum-wt-miner-ui`, 83a293f and 364feef, then the commit its agent reports after resolving it against gpu-hotplug's card-row states) | merged LAST, after F, at the agent's final commit; the node stays 21d4c73c, so the app, the DMG, CI and the ship files are rebuilt (section 7a) |
| H. The pack loader fix (the outage of 18:31Z, root cause found by the coordinator's agent): the generator retries a rejected candidate program with `seed || k` (epoch 34's attempt 0 was rejected on the saturation rule, attempt 1 accepted) and the workers' pack loader (`proto-cuda/nvrtc/packfile.h`, shared by the OpenCL worker) derived the expected seed words from attempt 0, so every worker started after the boundary refused a correct pack. Changes `packfile.h`, `worker.cpp`, `proto-opencl/host.c`, the relay's parse and tests, an emu test. It changes the Windows WORKER binaries: since 0.3.6 the payload has carried only the worker sources (`host.cu`, `host.c`, `build.bat`) and the PCs built the workers themselves, so the fix reaches the PCs only as exes: both workers cross-built on this Mac (`proto-cuda/nvrtc/build-windows.sh`, mingw, the staged redist) and carried by `push-inputs.sh` into the installer; a hot-fix copy is already on both PCs under `packs\\workers-fix`. The CI builds no worker (`windows.yml` copies the two worker exes and the `nvrtc*.dll` files from the signed inputs only), so the rebuilt pair reaches the installer through `push-inputs.sh` and nowhere else; the engine takes a bundled worker before its own PC-built one (`detect.rs` `Bins`, `engine.rs` `build_worker_from_source` only "when no prebuilt worker ships"), so the fix applies at the first start after the update | main `pack-loop` af983a7 (`igneum-wt-pack-loop`; the tip at merge time) | merged after G |
| I. The export lock: `engine.rs` serialises `igneum-miner export-pack` behind `EXPORT_LOCK` around both call sites (two per-card export threads interleaved into one folder across an epoch change on PC 1, a second way to a wrong pack) | main `opencl-rdna4` a08c371: the lock hunk is the part that must ship; the rest (host.c's duplicate-platform fold, `--readback select`, `--memprobe`, `detect.rs parse_opencl_list`) only if it merges cleanly with hotplug and the OpenCL worker cross-builds without error, else the hunk alone (coordinator, 19:2xZ) | (decided at the merge, section 2) |
| NOT in 0.3.10 unless the coordinator says so: `job-console` d33a266 (one hidden-console builder for every elevated launch, the spawn check in CI, PC 1 console watchers; its agent's message at 18:5xZ) and `opencl-rdna4` a08c371 (its agent's message at 19:1xZ: the OpenCL worker folds the duplicate AMD platform out of `--list`, a GPU select pass, `--memprobe`, `detect.rs` parsing with tests, and `engine.rs` serialises `export-pack` behind `EXPORT_LOCK` because PC 1's `packs\devnet` has held a mismatched `program.h` and `seeds.txt` since 19:07Z: the pack-export class the rollout is held for; passed to the coordinator as a candidate) | the coordinator's rule: a late branch goes in only if it lands before the final tree; the final tree b958493 was pushed at 18:36Z with CI green at 18:42Z. job-console also moves elevated-exit's exit-251 launcher text to `platform.rs`, so its `include_str!` test is repointed at its own merge to master after 0.3.10 | next cut |
| The app engine otherwise | unchanged except the version (0.3.10 in the six files); the signed jobs envelope client is master's (housekeeping C, 318a2de) | |
Changelog line for G (the coordinator's words): "The miner is now six sections: Mine, Prove, Rewards, Node, Updates, Settings. One switch per card, every setting with one line of help, the node's next switch shown with its height."
Changelog line for F (the coordinator's words): "A remote job that needs administrator rights now reports when the prompt was not accepted instead of claiming success."
Changelog line for E (the coordinator's words): "New cards are picked up while the miner runs; a card that goes away is released; integrated GPUs stay off unless you switch them on.
| The pinned guest | UNCHANGED unless C or D touches `proving/igneum-prove/core` or `elf/` (checked at the merge: section 2) | |
## 2. The branch
| Commit | What |
|---|---|
| 9268b64 | `origin/master` at the worktree's creation (18:00Z: the 0.3.9 merge, fast-forwarded) |
| 2e943a8 | Merge `tx-gossip` (0166e94): the relay design note, rows 463 and 9, the 3-node relay harness and its evidence; no conflict |
| 824059b | `Igneum Miner 0.3.10: the six version files` (`node tools/ship-app.mjs --check`: 0.3.10 in all 6) |
| 9997ef1 | `packaging/mac/packaged-config.sh`: the four-field override object in the packaged line, the self-test fix (section 3) |
| 0cbaf3a | Merge `c4-fix` (3072919, 18:20Z). One conflict, `docs/bench-log.md`: both entries kept (the fee-table entry, then the C4 entry). Brings `tools/ci/signer-pipe-check.sh` (`igneum-ota-sign ... \| head -1` under pipefail panicked the signer with SIGPIPE on a loaded Mac and killed a publish; now `sed -n 1p` in `publish-jobs.sh`, `publish-manifest.sh`, `push-inputs.sh`) and one `build-inputs-<time>-<pid>.zip` per build job (`build-job.mjs`, `push-build-inputs.sh --name`: with the shared name a job pinned another agent's sources three times tonight). From here every PC job and publish runs from THIS worktree's tools: they carry master's signed envelope (housekeeping C) and these two fixes, which the main checkout lacks until this branch merges |
| 4d10c20 | Merge `gpu-hotplug` at its tip at merge time, 4d122e1 (the agent's second commit: one row per physical GPU across OpenCL platforms, Windows names on the rows, index-free card keys, the OpenCL worker's `--list` prints the PCI address in `proto-opencl/host.c`); no conflict, 15 files |
| 5520fb1 | Merge `elevated-exit` (9816d58); no conflict |
| d1f4923 | Merge `origin/master` a93199a (docs and site only) |
| e0a5fd1 | `infra/cross/build-linux.sh`: the three paths made absolute before the cd (section 3) |
| 065c67c | `packaging/windows/node-source.pin` = 21d4c73c (push-inputs, section 5c) |
| b958493, ccd2b98 | the site as the pre-push hook builds it; the plan so far. The FIRST final tree: pushed 18:36:44Z, CI green 18:42Z, installer and DMG built (section 7). Then the project lead's scope change reopened it for G |
| 7f9618a | Merge `miner-ui-2` at its agent's final commit a3d9f2d (19:2xZ; it already carries gpu-hotplug 4d122e1 merged and resolved onto the new UI). One conflict, `packaging/windows/push-build-inputs.sh`: the usage comment only (c4's `--name` text against miner-ui-2's `--no-node` text; both options were already in the code), both kept |
| 5abc709 | Merge `pack-loop` (af983a7, its tip at merge time). One conflict, `relay/test/parse.test.mjs`: the import line (the union: `parseAppTail`, `parseCardsLine` from hotplug, `PACK_MISMATCH` from pack-loop) and two tests that both landed at the file's end (hotplug's cards line, pack-loop's pack mismatch), both kept; `node --test relay/test/parse.test.mjs`: 7 pass |
| 854f9a8 | `engine: one pack export at a time`: the `EXPORT_LOCK` hunk of opencl-rdna4 a08c371 (`git diff origin/master...a08c371 -- app/igneum-app/src/engine.rs`, applied clean: 9 lines, a static mutex and a guard at both export-pack call sites). The rest of that branch conflicts with hotplug in `detect.rs` and `proto-opencl/host.c` (a dry-run merge at 19:1xZ), so by the coordinator's rule it waits for the next cut |
| 854f9a8 | the fast checks on the SECOND final tree, 19:24Z: identity 0 hits over 213 files, copied-sources, pinned-guests, signer-pipe-check ok, no-conflict-markers ok, workflow shell 0 findings, `--check` 0.3.10 in all 6, relay tests 17 pass, UI tests 23 pass (notices 12, update-card 6, view 5; `view.test.mjs` in ci.yml line 85) |
| 5520fb1 | the fast checks on the first final tree: identity 0 hits over 212 files, copied-sources, pinned-guests, signer-pipe-check ok, no-conflict-markers ok, workflow shell 0 findings, `--check` 0.3.10 in all 6, relay tests 16 pass (hotplug's parse test added), UI tests 12 pass; 18:21Z |
A `vendor` symlink to the main checkout's `vendor/` (untracked) makes the relative node paths resolve, as in the earlier cuts.
The app and prover target dirs were cloned by APFS from the 0.3.9 worktree (18:01Z).
The fork branch `release-0.3.10` in `vendor/igneum-node-0310`, from a24ab01a: 2f88a82f (housekeeping 4fb32865), 21babaa7 (tx-gossip e242acd0), 21d4c73c (c4-fix e18f1e0e: `finality.rs`, `blockrelay/flow.rs` and five small files, 601 insertions), all three without conflict. Neither c4-fix nor tx-gossip touches a params file, `proving/igneum-prove/core` or `elf/`: the pinned guest and the prover rollout are untouched by 0.3.10.
## 3. Tests and checks, with the command
| Tip | Command | Result |
|---|---|---|
| fork 2f88a82f (a24ab01a + housekeeping) | PC 2 job `build-20261005-164604` (`node tools/build-job.mjs run --node vendor/igneum-node-0310 --target 1ccfe586 --targets linux --node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner" --no-app`) | started 16:46Z, done 16:48:43Z (129 s): Linux node built, `RESULT test node [the five packages] exit 0 39 s`: the parallel five-package run is green on the housekeeping tree (housekeeping B); the short report carries the per-package exit, not the per-test names |
| fork 2f88a82f | PC 1 job `build-20261005-164512` (`--targets linux,windows --no-tests --no-app --no-place`) | started 16:45Z, done 16:50:26Z (292 s), every stage ok, 7 files, all sha256 and PE checks ok: igneumd.exe 37d0b1045043bc86... (50,889,728), igneum-miner.exe bc8cb3a12111e7bb... (10,725,888), Linux igneumd d0331c109babe77e... (48,733,736, glibc 2.39: HiveOS, not the seed). `strings igneumd.exe`: msvcrt.dll is the only C runtime named; no libstdc++-6, libgcc_s_seh-1 or libwinpthread-1 (housekeeping A) |
| fork 2f88a82f, PC 1 exe | run job `run-0310-accept-hk` (the housekeeping agent's acceptance script: rocks-probe, igneumd 60 s on a scratch appdir with no mingw DLL beside it, `igneum-miner.exe key-hash probe`, Application event 1000 check), published 16:51:01Z (the apps woken) | ran 16:57:15 to 16:58:21Z (66 s), exit 0: `rocks-probe` exit 0 (21 files in its db), `RESULT run node60: still running after 60 s (alive for the whole wait); killing it` with 14 files written under the scratch appdir, `igneum-miner.exe key-hash probe` exit 0, `RESULT event 1000: none`; the exe's imports on the PC (objdump): KERNEL32, advapi32, bcrypt, iphlpapi, msvcrt, ntdll, ole32, ws2_32 and the api-ms-win-core set, no mingw DLL. The one warning, `cannot bind the eth_ JSON-RPC server on 127.0.0.1:26790`, is PC 1's live node holding the port, as in the housekeeping run. So the PC-built Windows node is the 0.3.10 Windows node (no Mac cross-build needed) |
| fork 2f88a82f, Mac arm64 | `CARGO_TARGET_DIR=vendor/igneum-node/target-0310 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` from `vendor/igneum-node-0310` under the lock (target dir cloned by APFS from `target-036`; a new worktree path rebuilds every crate) | 16:44:52 to 16:57:28Z (12 min 36 s): igneumd 2a7df6245ffe047a... (40,968,368, `2f88a82f` in its strings), igneum-miner 322472ae95a316d1... (8,514,928) |
| fork 2f88a82f, Mac arm64 | the digest check: that igneumd on a scratch node (ports 60975/60976, 22 s, 16:57:58Z) with `{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000}` | `Calibrated v1 fees from the override file: ... from DAA score 210000`, digest ab8847da538dead1dc10e046dfaadab3c1c35928e3748810c4e050d4a886087a, `igneumd/2.1.0-2f88a82f`: housekeeping moves no parameter |
| fork 2f88a82f, Linux x86-64 for the seed | `NODE_SRC=vendor/igneum-node-0310 TARGET_DIR=vendor/igneum-node/target-0310-linux infra/cross/build-linux.sh` (zig, glibc 2.36 target) under the lock | queued 16:57Z behind two other agents' builds, built 17:23 to 17:35:49Z (754 s, 564 crates), and its output was WRONG: igneumd 7e26e374... BYTE-IDENTICAL to the 0.3.9 build of a24ab01a, embedding `a24ab01a`. Cause (found at the second run, 18:25Z): `build-linux.sh` does `cd "$NODE_SRC"` and runs cargo with `CARGO_TARGET_DIR="$TARGET_DIR"`, so a RELATIVE target dir resolved inside the node worktree (`vendor/igneum-node-0310/vendor/igneum-node/target-0310-linux`, the untracked `vendor/` that appeared there), while the copy step read the repository-relative clone that still held the a24ab01a binary. Every crate had compiled, into the wrong place. Fixed on this branch (e0a5fd1: the three paths made absolute before the cd) and the build rerun with an absolute target (section 5). My first reading of it, a stale `kaspa-build-info` output in the cloned target, was wrong and is withdrawn |
| fork 21babaa7 (2f88a82f + tx-gossip e242acd0), Mac arm64 | the same cargo build, incremental on `target-0310` | 17:49:52 to 17:52:00Z (2 min 08 s): igneumd 5d7f1b576029b462... (41,118,944, `21babaa7` in its strings) |
| fork 21babaa7, Mac arm64 | the digest check as above (ports 60975/60976, 22 s, 17:52:01Z) | `Calibrated v1 fees ... from DAA score 210000`, digest ab8847da538dead1dc10e046dfaadab3c1c35928e3748810c4e050d4a886087a, `igneumd/2.1.0-21babaa7`: tx-gossip moves no parameter (its commit message says the protocol version is not in the digest; measured here) |
| fork 21babaa7 | PC 2 job `build-20261005-175022` (`--node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows" --no-app`) | started 17:50Z, done 17:53:10Z (145 s): `RESULT test node [the six packages] exit 0 53 s` (kaspa-p2p-flows added: tx-gossip's relay tests live there and in igneum-exec) |
| 2e943a8 + bump | `tools/ci/identity-check.sh` (0 hits over 212 files), `copied-sources-check.sh`, `pinned-guests-check.sh` (elf/ matches its manifest), `node tools/ci/check-workflow-shell.mjs` (12 run blocks, 25 .ps1, 0 findings), the relay tests (15 pass), the UI tests (10 pass) | all green, 18:00Z |
| fork 21babaa7, Mac arm64 | the digest check with the FOUR-field object `{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}` (N3 = 135,200 from the live 0.3.9 manifest republished 17:59:14Z, deadline note "finality v3"; the project lead: "deploy N3 now", the 0.3.9 agent publishing it) | 18:01:59Z, 22 s: `Finality rule v3 from the override file: active from checkpoint DAA score 135200`, `Calibrated v1 fees ... 210000`, digest 1f4b44255fcd2ea8f75664ed47200f409186ddd2292960c9e2cf95bbbdc11505, `igneumd/2.1.0-21babaa7`. Equal to the live hand nodes' lines at 18:02Z (`observer-v4.out`, `node1.out`: digest 1f4b4425..., `igneumd/2.1.0-a24ab01a`, the four-field `/tmp/igneum-devnet/override-v3.json`): the 0.3.9 agent's N3 rollout had reached them. This is the digest every node must show; the final 0.3.10 node is read again against it |
| 9997ef1 | `packaging/mac/packaged-config.sh`: `NODE_OVERRIDE_PARAMS` = that four-field object (it was empty on master; the manifest's `consensus.override` is what the apps write, the packaged line covers a first start before any manifest). Its self-test's "json has no override line" check read the file's shipped default and failed on a non-empty line (master's empty line passed it), so that check now passes `NODE_OVERRIDE_PARAMS=''` itself, as the "override params land" check already passes its own value; `bash packaging/mac/packaged-config.sh --test`: all checks passed |
| fork 21d4c73c (+ c4-fix), app 5520fb1 | PC 2 job `build-20261005-182244` (`--node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows" --app-tests igneum-app`) | 18:23:15 to 18:25:41Z: the node stage FAILED, exit 101 after 44 s: `kaspa-consensus` 96 passed, 1 failed, 3 ignored: `processes::finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list` panicked at `finality.rs:1883:82` (the `mine_on_all` helper's `validate_and_insert_block(...).unwrap()`) with `UnexpectedDifficulty(<bits>, 487112096, 487129281)`: a block built on one `TestConsensus` was refused by another for a difficulty a few hundred bits off. cargo stopped at the first failing package, so the other five were not run in this job. The app tests passed: `igneum-app` exit 0 in 6 s (hotplug's and elevated-exit's tests included; 96 + the signer and wrapper suites) |
| fork 21d4c73c | PC 2 job `build-20261005-182804` (`--node-tests kaspa-consensus`, the suite ALONE) | 18:28:3x to 18:30:15Z: `kaspa-consensus` 97 passed, 0 failed, 3 ignored in 1.99 s, `ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list ... ok`. So the test passes alone (as in the c4 agent's own run, build-20261005-180827, 97 passed) and failed once under the six-package parallel run: the 0.3.6 isolation class (release-0.3.6 plan 8f), now with a difficulty expectation that depends on wall-clock timing rather than the PoW cache queue. Not a consensus regression by the same reasoning as then; recorded in section 11 for the c4 agent (the test or the helper should pin its clock) |
| fork 21d4c73c | PC 2 job `build-20261005-183131` (`--node-tests "kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows"`) | 18:31:31 to 18:33:59Z: exit 0 in 37 s: `kaspa-consensus-core` 101 passed (2 ignored), `igneum-exec` 15 passed, `igneum-miner` 15 passed, `kaspa-p2p-flows` 33 passed (the target compiles and passes on the merged tree: the nine `epoch_seed_headers` errors of release-0.3.6 are gone with tx-gossip's `ibd/proof.rs` update, as the c4 agent said), `kaspa-pow` 13 passed (both queue tests of housekeeping B). With `build-20261005-182804` (consensus alone, 97 passed) every package of the six is green on 21d4c73c; the app suite passed in `build-20261005-182244` |
| tree 9a51e32+ (before pack-loop) | `proto-cuda/nvrtc/build-windows.sh` under the lock (mingw-w64 from Homebrew, the redist dir of the main checkout symlinked into the worktree: NVRTC 12.8.93 DLLs and headers, Khronos CL headers, nothing downloaded) | 19:17 to 19:18:17Z, a trial of the script before the pack-loop merge: igneum-worker-cuda.exe 196082e2... (1,509,376) and igneum-worker-opencl.exe 0981c77d... (444,928), both with the Igneum resource block (whose version string is 0.3.0: the `.rc` files are not among the six version files, section 11); the real pair is rebuilt from the final tree (section 5d) |
## 4. The state of the network before the cut
Both of the 0.3.9 agent's rollouts finished before anything of 0.3.10 moved: the fee switch (H = 210,000, every node on
ab8847da..., `release-0.3.9.md` 10e, 17:55Z) and then N3 (the project lead: "deploy N3 now"; `finality_v3_activation_daa` 135,200,
manifest republished 17:59:14Z, `docs/plans/finality-v3-devnet-publish.md`): the sweep there lists every live node on
`1f4b44255fcd2ea8f75664ed47200f409186ddd2292960c9e2cf95bbbdc11505` (the observer 17:58:13Z, node 1 17:58:25Z, the seed
17:58:43Z, PC 2's app node 18:04:57Z, PC 1's app node 18:06:28Z). The override object every node runs, and the one the
0.3.10 manifest and packaged line carry:
```
{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200}
```
| Node | Binary at 18:03Z | Override | Digest |
|---|---|---|---|
| Mac d937c69d | app 0.3.9; its engine has run NO node of its own since its hourly restart at 17:45:16Z: it found node 1 answering on 127.0.0.1:26610 and uses it (the N3 record, "the Mac app's external-node finding"), so the Mac card's node line is node 1's | node 1's | node 1's |
| PC 1 ae432dc7, PC 2 1ccfe586 | app 0.3.9, node a24ab01a (the Mac cross-build) | the four-field object | 1f4b4425... |
| Sam's Mac 3a9bf309 | app 0.3.9, node a24ab01a, 0.0 MH/s at 18:03Z | (its app writes the manifest's object) | not read |
| The observer, node 1 (hand nodes, this Mac) | `vendor/igneum-node-036/target-integration/release/igneumd` a24ab01a | `/tmp/igneum-devnet/override-v3.json`, the four-field object | 1f4b4425... (read 18:02Z) |
| The seed 188.245.5.161 | `/opt/igneum/v4/bin/igneumd` 7e26e374... (the a24ab01a zig cross-build, glibc 2.34) | `/etc/igneum/override-v3.json`, the same | 1f4b4425... |
| PC 37ba0461 | silent (0.3.7 at the 0.3.8 cut) | | |
Master moved under the branch while it was cut (fef98a2/a850aef CLAUDE.md, 900bba7 the site's 0.3.9 download cards, 46a6a4e and
bf8d1ba the N3 record, a93199a); merged as d1f4923 (docs and site only, 5 files, no conflict). It moved again at 19:4xZ (the coordinator's
`explorer` merge, origin/master 9746391: the observer's explorer detail, `site/api/stats.mjs` and `supply.mjs`, the explorer pages, three node
tests and `tools/ci/public-api-check.mjs` in ci.yml; no app, node or packaging file): the final merge to master lands on it (section 7b).
## 5. The node binaries and the DMG (fork 21d4c73c, app 5520fb1 and later)
| Platform | Build | sha256 | Size | Notes |
|---|---|---|---|---|
| Mac arm64 igneumd | `CARGO_TARGET_DIR=vendor/igneum-node/target-0310 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` from `vendor/igneum-node-0310`, under the lock, 18:21:54 to 18:24:2xZ (incremental) | 4bb356f492a52154c932ee6f802cd59dc5c917840ad46ba2133905f52b838b3c | 41,152,144 | `21d4c73c` in its strings; copied into `vendor/igneum-node-0310/target-integration/release/` |
| Mac arm64 igneum-miner | same | 322472ae95a316d12281fed99b2112cdb5f7b6caf9f25ea24019219b7e907948 | 8,514,928 | byte-identical to the 2f88a82f build: the miner is untouched by gossip and c4 |
| The digest check | that igneumd on a scratch node, ports 60975/60976, 22 s, 18:24:23Z, the four-field object | `Finality rule v3 ... 135200`, `Calibrated v1 fees ... 210000`, digest 1f4b44255fcd2ea8f75664ed47200f409186ddd2292960c9e2cf95bbbdc11505, `igneumd/2.1.0-21d4c73c` | | equal to every live node's (section 4): the three fork branches move no parameter; the chain script stops on any other value |
| Linux x86-64 igneumd (the seed, glibc 2.36) | `NODE_SRC=<abs> TARGET_DIR=<abs> OUT_DIR=<abs> infra/cross/build-linux.sh` (zig), under the lock, 18:25:43 to 18:27:29Z (105 s incremental) | 40be0e142e30f6de89ccb61f6dc13e8748ed2a41988753604dbdbb9da1466b4b | 47,638,184 | `GLIBC_2.34` at most, `21d4c73c` in its strings; `version.txt` from 21d4c73c (release-0.3.10). Kept in the scratchpad `r0310/cross-final2/` and copied to `infra/cross/out-0310/` of the main checkout for `restart-seed.sh` |
| Linux x86-64 igneum-miner | same | 6ec3536717536fd505b63eef5d01ec9fa19d1fe608ba86398135c5776222ec73 | 9,631,280 | |
| Windows x86-64 igneumd.exe | PC 1 job `build-20261005-182405` (`IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release node tools/build-job.mjs run --node vendor/igneum-node-0310 --target ae432dc7 --targets linux,windows --no-tests` from this worktree: its own `build-inputs-20261005182334-74461.zip`, the per-job name of 68f14b9), published 18:24:05Z, started 18:26:1xZ after another agent's job on PC 1, every stage ok 18:31:43Z (linux 136 s, windows 162 s, pack 3 s; 9 files, 45 MB, all sha256 and PE checks ok, placed) | 3adb01a712fba678706ed5eeb1fa9788895735e4fab5e81dbefe03bd675a5e04 | 50,978,816 | the PC build (housekeeping A); no commit string inside (section 11); acceptance: section 5b |
| Windows x86-64 igneum-miner.exe | same | 1105151c91738a4e177b8f327625c6e22a90423fe66a71a5d045130618281052 | 10,725,888 | |
| Windows x86-64 igneum-app.exe (the PC's build; the installer's engine is the GitHub runner's) | same | 4ead820eb4c502de18f8c122d319543452615ba96b8471a501eb2237e76835b7 | 2,925,056 | hotplug's first Windows compile (`detect::adapters`, the WM_DEVICECHANGE path is `host.cpp`, built by the runner): warnings only; among them hotplug's own `function adapter_for is never used` (for its agent), the rest pre-existing (94 `trailing semicolon in macro`, p2p-flows 21, dead `keep`/`wsl_path`) |
| Linux x86-64 igneumd (HiveOS, glibc 2.39, not the seed) | same, `infra/cross/out/` of this worktree | 38e69412b66a9ee183f3a28f25198f31ce29f83554f7a7f7dce9bdc4f61970be | 48,915,112 | |
| Linux x86-64 igneum-miner, igneum-app | same | 39efcb1773dc215d..., cd8680e5b8c07795... | 9,607,448; 2,370,544 | |
### 5b. The acceptance of the PC-built Windows node (housekeeping A)
Run job `run-0310-accept-final` (the housekeeping agent's `run-accept-pc1.ps1` unchanged: copies the three exes from PC 1's
`/root/igneum-build/target/x86_64-pc-windows-gnu/release` into a scratch folder with NO mingw DLL, runs `rocks-probe`, then
`igneumd.exe --devnet --nodnsseed --disable-upnp --rpclisten=127.0.0.1:26710 --listen=127.0.0.1:26711 --nologfiles --yes` on a scratch
appdir for 60 s, then `igneum-miner.exe key-hash probe`, then reads Application event 1000), published 18:32:37Z from this worktree:
ran 18:33:08 to 18:34:15Z (67 s), exit 0: `rocks-probe` exit 0, `RESULT run node60: still running after 60 s (alive for the whole wait); killing it`, `igneum-miner.exe key-hash probe` exit 0, `RESULT event 1000: none since 18:32:11Z`. The same verdict as the housekeeping run (`run-hk-accept-1`) and my warm-up run on 2f88a82f (`run-0310-accept-hk`, 16:57Z, the exe 37d0b104...). So the PC-built Windows node ships; no Mac cross-build was needed. The payload inputs: section 5c.
### 5a. The DMG
The chain under the lock (the 0.3.9 script's shape): the app (`cargo build --release` in `app/igneum-app`, 18:24:47Z, target cloned from the
0.3.9 worktree), the prover host and export (18:24:54Z, no Succinct toolchain: the host embeds `elf/`), `igneum-prove-host --mode id`
on this tree: shard program id `0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a` (2,832,504 bytes, sha256 0x150f4c05a2951fc5),
aggregator `0x474678f35f7545db28055d5e5bbc308231d84a5a072202087a2a8d5b09123896`: the 0.3.9 pin, unchanged (no prover rollout). The
nine real fixtures ran `--mode native` with the new host (338, 341, 344, 56 x2, 58927, 72803, 72854, 78: all ok); the first chain
stopped on `block-72803-skipped-copies.json.node-plan.json`, which is a node-plan SIDECAR of the 72803 fixture (txgen 49796b0), not a
fixture, so the glob now excludes `node-plan` sidecars. Then `NODE=<fork>/target-integration/release/igneumd MINER=... packaging/mac/build-dmg.sh`
(18:26:04 to 18:26:34Z): `prover: ... from <worktree>/proving/igneum-prove/target/release`, the same `--mode id` line, fingerprints 477bb0ef
(intake key) and ed9c4d2e (folder).
| Artefact | sha256 | Size |
|---|---|---|
| `packaging/mac/dist/Igneum-Miner-0.3.10.dmg` (engine 0.3.10, node 21d4c73c Mac arm64, the 0.3.9 prover host and export rebuilt from this tree) | e8f1f9f620cb598fd21ff2b6f6a2dea3d9e58efc0550c6454dc65f0a3d791b97 | 41,377,608 |
Read back from the DMG (mounted read-only, 18:28Z): `Contents/Resources/igneum-app.json` carries `node_override_params` =
`{"difficulty_v2_activation_daa": 33000, "fees_v1_activation_daa": 210000, "finality_v3_activation_daa": 135200, "proving_v0_activation_daa": 84100}`
(the packaged line of 9997ef1) beside `update_manifest`, `log_intake_url`, `log_intake_key`, `live_page`, `download_page`; `bin/` holds
igneumd, igneum-miner, igneum-bench, igneum-prove-host, igneum-prove-export.
A note on my own logs: the chained scripts print `<time> ... exit $?` lines where the time is a command substitution evaluated first, so
those lines always say 0. Every outcome in this plan is read from a content line (`Finished`, a sha256, a `RESULT`, a `final:`), never
from them; `tools/lock/with-lock.sh` itself propagates the exit code (checked: `run bash -c 'exit 3'` returns 3).
## 6. The harness runs (C and D)
| Run | Node | Command | Result |
|---|---|---|---|
| D, early (before c4) | fork 21babaa7 Mac arm64 (`target-0310`) | `IGNEUMD=... IGNEUM_MINER=... IGNEUM_HARNESS_BASE_PORT=29750 IGNEUM_HARNESS_TMP=/tmp/igneum-txrelay-0310 tools/lock/with-lock.sh run node tools/txgen/relay-net.mjs --rate 2 --duration 120 --wallets 16 --fund 2` from the `tx-gossip` worktree (the harness was not yet on this branch) | 17:54:26 to 18:01:01Z: `final: sent 240, included 240, pending 0, failures 0`, 1.951/s, latency p50 2,017 ms, p90 3,561 ms, max 6,560 ms: the coordinator's expected 240 of 240 |
| D, final | fork 21d4c73c Mac arm64 (igneumd 4bb356f4...) | the same command from the `tx-gossip` worktree (this worktree has no `node_modules`: the harness imports `viem`, the first attempt from here died at import; `tools/txgen/` needs an install note), ports 29750+, data `/tmp/igneum-txrelay-0310b` | 18:26 to 18:31:05Z: `final: sent 240, included 240, pending 0, failures 0`, 1.975/s, latency p50 1,537 ms, p90 3,027 ms, max 5,025 ms; report in the scratchpad `r0310/relay-final2/` (by miner: B and C only, A none, as the harness requires) |
| C, first run | fork 21d4c73c Mac arm64 (igneumd 4bb356f4...) | `node tools/finality-attacks/c4.mjs on` with the script's DEFAULTS (split 150 s, weight window 120 DAA; ports 29900+, suffix 990, `/tmp/igneum-fin-c4-0310`), 18:24:50 to 18:34:43Z | `[FAIL]`: B locked 7 and 8 during the split, n0 reconnected 3 s after the heal, the three sinks DISAGREE, 0 locks adopted from an off-chain certificate, 9 / 4 / 4 conflicting certificates. The same shape as the bench-log's "on, split 150 s" row on the fix: a 150-s split is past the 120-DAA frozen table, so by the heal A's chain has outrun the table and the certified chain cannot be adopted; the c4 agent's measured PASS rows use `WINDOW=240` (a 140-s split stays inside the 126-DAA bound, as any partition under an hour does on the live devnet) |
| C, final | the same binary | `WINDOW=240 SPLIT=140 node tools/finality-attacks/c4.mjs on` (the bench-log's PASS row's knobs, `/tmp/igneum-fin-c4-0310b`), 18:36:3x to 18:46:20Z | `[PASS] weight-vs-work-on`: B locked index 9 during the split (138 s after the cut), A's sink blue score 315 against B's 292 (A the heavier by work), n0 reconnected 3 s after the heal, the three sinks AGREE (0424542743) on B's chain, A's split tip abandoned, n0 adopted 1 lock from a certificate off its chain (the C4 fix) and re-determined 1 line, 0 conflicting certificates, 0 reorg lines, max locked index 15 on all three. The coordinator's rollout note holds on the shipped binary: the certified chain wins over the heavier one when every node is new |
### 5c. The Windows payload inputs (18:35:33 to 18:36Z)
`IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release IGNEUM_NODE_SRC=<fork> packaging/windows/push-inputs.sh` from this
worktree (the `sed -n 1p` signer read of 68f14b9): igneumd.exe 3adb01a7... (50,978,816) and igneum-miner.exe 1105151c... (10,725,888), the PC 1
build of 21d4c73c, with the PC's three mingw DLLs (libstdc++-6 26,347,027, libgcc_s_seh-1 774,200, libwinpthread-1 324,451; the exes import
none of them since housekeeping A, the DLL gate and the copy stay for an older fork). Signed inputs: zip
c175446e33ccb020e2f5fdaf99100d5bd46138c7693b3e16948a82f7c252eaa4 (33,996,479 bytes, 5 files), `node_source_commit`
21d4c73c6ce32fcbd68391968e85827339a511c0 (release-0.3.10), `repo_commit` 9a51e32, built 18:35:37Z, deployed, `payload-inputs.json` HTTP 200;
`node-source.pin` = 21d4c73c, committed as 065c67c.
## 7. The ship
`git push -u origin release-0.3.10` at b958493 (18:36:44Z, the credential helper; `gh auth switch --user igneum-labs` before every gh call,
the login renamed igneum-labs at 18:00Z, the token still under that name), `gh workflow run windows.yml --ref release-0.3.10` -> run
37357266092 (queued 18:36:49Z on b958493).
The pre-push hook (fb076de: no conflict markers, `cd site && node build.mjs`) left `site/index.html` and `site/journey.json` modified after the
push. Measured with three builds of one tree at 18:38Z: the label of one bench entry ("RTX 5090 first run" against "RTX 5090, memory-hard
dataset") ALTERNATES on every run, whatever the cwd: `build.mjs` reads `site/journey.json` back (line 248) and writes it, so each run
transforms the previous output. Recorded in section 11. The hook's output (master's form) is committed as ccd2b98 and the next push's
hook flipped it again; the tree is restored with `git checkout -- site/` after every push, so the ship's preflight sees it clean. Those commits touch no
app or packaging file, so the installer built at b958493 is the release; the ship state file (`~/.cache/igneum/ship/0.3.10.json`) got `sha`
= b958493 and `runId` = 37357266092, what the tool's own steps would have recorded, and the run resumes with `--from ci`.
Run 37357266092: green at 18:42:18Z (parse checks 52 s; engine, window host, payload, installer, smoke run 4 min 24 s; the G13 inputs step
against the 21d4c73c inputs of 5c and the pin of 065c67c). The dry run of the ship command (18:38:54Z) read everything: tree ccd2b98 clean,
0.3.10 in all 6, fork 21d4c73c, both exes, both Mac binaries, gh igneum-labs, live inputs 21d4c73c built 18:35:37Z, live manifest 0.3.9.
HOLD (coordinator, 18:4xZ): both PCs' NVIDIA workers have looped since 18:31Z (PC 1) and 18:35Z (PC 2) on `the epoch seed bytes do not give
the pack's IGNEUM_SEEDW_INIT` (the exported GPU pack is the previous epoch's and the restart loop never re-exports it; the network fell to
about 88 MH/s); the coordinator published re-export jobs and holds every rollout until the fleet mines again. The ship steps that deploy
nothing ran by hand and the manifest step waits: `OTA_SKIP=1 CONSOLE_SKIP=1 packaging/windows/fetch-ci-artifacts.sh 37357266092` (18:43:26Z)
and the DMG copied into the downloads folder.
| File (in the downloads folder, not yet deployed) | sha256 | Size |
|---|---|---|
| Igneum-Miner-0.3.10.dmg | e8f1f9f620cb598fd21ff2b6f6a2dea3d9e58efc0550c6454dc65f0a3d791b97 | 41,377,608 |
| Igneum-Miner-Setup-0.3.10.exe | 7ecf109f3867b554ecb97be3542f23c8d76d9752d6bdc7ece7a6f95fc2f9bfd4 | 24,994,420 |
| igneum-windows-app.zip | ebf0deca0e99de4085af1af5c3349028d8a8386f5016ee357ecf806df9a329c9 | 34,778,147 |
The installer is 5.2 MB larger than 0.3.9's (19,836,912) and the zip 7.1 MB larger: the PC's `libstdc++-6.dll` is 26,347,027 bytes
(Ubuntu's GCC 13 build, unstripped) against the Mac toolchain's, and it rides in the payload although the exes import no mingw DLL since
housekeeping A (section 11).
The ship command, to run when the hold lifts (every step before `manifest` skips itself):
```
node tools/ship-app.mjs 0.3.10 --node vendor/igneum-node-0310 --branch release-0.3.10 --public \
--activation-height 135200 --deadline-note "finality v3" --notes "<the five changelog lines, section 1; node 21d4c73c>" --from ci
```
`consensus.override` is carried over from the folder's 0.3.9 manifest (the four-field object of section 4) and `--activation-height 135200
--deadline-note "finality v3"` keep the consensus object identical field by field to the live one.
### 7a. The second final tree (the project lead's scope change, 19:0xZ): the app rebuilt
The node stays 21d4c73c (its binaries, the digest, the suites, the harness runs and the Windows node acceptance all stand). What rebuilds:
the app (G, H, I: the UI, the engine additions, the export lock), the two Windows WORKER exes (H changes `packfile.h`, `worker.cpp` and
`host.c`), the payload inputs (now carrying the workers and the NVRTC DLLs), the DMG, the installer (CI), and the ship files.
| What | Result |
|---|---|
| The two Windows workers, `proto-cuda/nvrtc/build-windows.sh` under the lock, 19:24Z, from 854f9a8 | igneum-worker-cuda.exe 85cc357bb62bda36634ff4c1d80aca575aad71204ff2ef4bf620bdf183ab9236 (1,510,912), igneum-worker-opencl.exe afa73a324b9bfc3d957d0d5d3b6453770046721ac40cf7455ff3d49eb8b5774f (445,952); both differ from 0.3.9's (f50e19f2..., 1abc673e...) and from the hot-fix pair on the PCs (8dcef61c..., af7f12f5...), as the coordinator's check requires; the resource block on both |
| `push-inputs.sh` again, 19:24:41Z | 12 files: the node pair and three DLLs of 5c, the two workers, nvrtc64_120_0.dll (86,728,192) and nvrtc-builtins64_128.dll (6,356,480), the three licence texts; zip ec1af603cb047471f8e6d6d95adbad192e0b77095e031fd4e39a807e37e8bc5c (71,940,402 bytes), node 21d4c73c, repo 854f9a8, signed, deployed, HTTP 200; the pin unchanged |
| `cargo build --release` in `app/igneum-app`, then `cargo test --release -p igneum-app`, under the lock on this Mac (the coordinator's ask) | 19:24Z: ok, 103 (lib) + 27 (ota-sign) + 8 (prove-verify), 0 failed; `igneum-app 0.3.10` |
| `--mode id` on the worktree's host (unchanged prover) | shard `0x2b1a81cb...`, aggregator `0x474678f3...` |
| The DMG, `packaging/mac/build-dmg.sh` under the lock, 19:24:57 to 19:25:18Z | `Igneum-Miner-0.3.10.dmg` 151687c5672553c571af7224a8e4228348135b8a313fc89e6641db7ae21c85b3, 41,383,375 bytes (engine 0.3.10 with G, H, I; node 21d4c73c; the 0.3.9 prover), fingerprints 477bb0ef and ed9c4d2e; it replaces the e8f1f9f6... DMG of the first tree |
| PC 1 app-only job (`node tools/build-job.mjs run --target ae432dc7 --targets windows --no-node --no-tests`, the branch's new `--no-node`) | published 19:24:50Z, done 19:26:47Z in 14 s (the engine alone, 6 s on PC 1's warm cache): igneum-app.exe 4564f8502bf40fdfb61a979ff19543838ca66be95b6b9f0beeaee53330c4d284 (2,962,944), warnings only (the dead-code set of 5a plus `count` is never used); `host.cpp` (WM_GETMINMAXINFO, WM_DEVICECHANGE) compiles on the runner only: CI run 37363381420 (section 7b) |
Not in this tree (they arrived after it; next cut): `job-console` 3562f26 (its second offer, 19:2xZ: power control as a setting, default off), and
the FORK-side `pack-loop` 05ef0fa3 (`vendor/igneum-node`: the miner checks every pack and rebuilds a refused one, exit 44; a node change: the
node stays 21d4c73c). The app side of af983a7 (`watchdog.rs` `PackRebuilds`, the engine re-exporting the pack on a refusal, capped per
epoch) is keyed on the miner's exit 44, which only the fork-side miner emits (the pack-loop agent's correction, 19:3xZ): with the 21d4c73c miner a worker refusal line shows "program pack out of date, rebuilding" on the strip, the card and the log but does NOT re-export; with the attempt-aware workers of this release a refusal means a genuinely broken pack, so the gap is small, and the self-healing half lands with the next node cut (section 11).
### 7b. The second push and CI
`git push origin release-0.3.10` at 5b0d54f (19:26:13Z; the pre-push hook flipped the two site files again, restored with `git checkout -- site/`),
`gh workflow run windows.yml --ref release-0.3.10` -> run 37363381420 (queued 19:26:19Z on 5b0d54f). The ship state file now names 5b0d54f
and that run. At 19:46Z the run was still QUEUED with no runner assigned, as were the repo's two `ci` runs (19:26Z, 19:39Z): GitHub's status
API reported Actions "degraded_performance" with an unresolved "Incident with Actions" (investigating since 19:15:17Z). Nothing in this
repository or on this Mac can shorten that: the installer comes only from the hosted `windows-latest` runner. The rollout waits for the
verdict; every other step is staged (the DMG, the signed inputs, the runbook `r0310/rollout.sh` with the exact commands).
Run 37363381420 ended at 19:41:24Z as `failure` with its first job CANCELLED by GitHub after 15 minutes queued ("The job was not acquired by
Runner of type hosted even after multiple attempts", the job's annotation) and the build job skipped: nothing of the tree ran. The workflow
was dispatched again under a watcher that dispatches again on that same annotation and stops on a green or on a failure of the tree itself:
try 2 run 37365130137 (19:42:43Z), try 3 run 37366744501 (19:57:52Z), try 4 run 37368355454 (20:13:21Z), try 5 run 37369931084
(20:28:32Z), every one cancelled by GitHub the same way after about 15 minutes queued; a follow-on watcher took over at 20:54:38Z and
dispatched try 6, run 37374158235 (20:54:55Z), which a runner acquired at 21:24:04Z and which went green at 21:30:29Z (section 7d).
Six dispatches, 2 h 04 min from the first to the green; GitHub's incident ("major outage" at 20:4xZ) was the whole of it.
### 7c. The fallback, prepared at 20:3xZ (the coordinator: a switch at 21:00Z, not a scramble; nothing built yet)
What the runner does for the installer, and what PC 1 has for each step (probe job `probe-installer-pc1-0310`, read-only, ran 20:33:5xZ
in 3 s; its `R` helper collided with PowerShell's `r` alias so every line came back inside an error message, the facts intact):
| Runner step | Needs | PC 1 (ae432dc7) | Fallback |
|---|---|---|---|
| engine (`cargo build --release --locked`, MSVC target) | Rust on Windows | not probed (the PC's build jobs build the engine in WSL on the GNU target: igneum-app.exe 4564f850..., 5a) | the GNU-target engine from job `build-20261005-192450` unless `cargo` exists on the Windows side (checked at the start of job B); recorded either way |
| packaged configuration | the intake key and the folder token as files | the installed 0.3.9 app's `igneum-app.json` at `C:\Users\Admin\AppData\Local\Programs\Igneum Miner\` carries the same manifest URL and key (`override=False`: 0.3.9 shipped no packaged override) | copy that file and add `node_override_params` = the four-field object (not a secret); no secret leaves the Mac |
| payload inputs (`payload-inputs.zip`, signature, hashes, node commit) | the dl folder URL | the URL's folder is in the packaged json | download, verify the sha256s against `payload-inputs.json` (the signature is verified by the runner's step with the key compiled into the app; on the PC the app's own `igneum-ota-sign` is not installed: recorded as a gap, the hashes stand) |
| window host (`app\windows\BUILD-APP.bat`, MSVC v143 cl.exe and rc.exe) | Visual Studio with MSVC | `vcvarsall.bat` under `C:\Program Files\Microsoft Visual Studio\...`, cl.exe 19.51.36260 | runs as on the runner |
| payload (`make-payload.sh`, bash) | Git Bash | `C:\Program Files\Git\bin\bash.exe` (and WSL) | runs as on the runner |
| installer (`build-installer.ps1`: Inno Setup 6 `ISCC.exe`, rcedit) | Inno Setup 6, rcedit-x64 | ISCC NOT FOUND; rcedit not on PATH; winget present | `build-installer.ps1` installs Inno Setup 6 through winget (a tool install on PC 1: the coordinator's call) and downloads rcedit-x64 from GitHub (`-NoRcedit` skips it: the exes then ship without the coin icon and version block, cosmetic, recorded) |
| smoke run, launcher dry run | the exes | | the same PowerShell lines in job B |
| the files back to the Mac | | `relay/clients/send.ps1` (a file up to 50 MB straight to Blob): the installer is 25 MB, the payload zip 35 MB | two `send.ps1 <file>` calls with the sha256s in the report; on the Mac `node tools/relay.mjs read <id>`, sha256 compared, `packaging/windows/check-runtime-dlls.sh` on the payload folder |
| code signing | none on the runner either | | none |
The jobs, in order (neither published until the word; both staged in the scratchpad `r0310/fb/`: the tree zip `fb-tree-0310.zip`, 709,814 bytes,
98 files from 5b0d54f by `git archive`, sha256 012c8a52c27631e8539704d703bf13eb68de0afd9cef094e68d1e6fc398f2cfc; the script `fb-installer-pc1.ps1`,
which stops on any sha mismatch, pins the inputs' node commit against `node-source.pin` as the runner's G13 step does, and takes `-NoRcedit` as its one argument):
1. `fetch` job `fb-tree-0310` to ae432dc7: a zip of the tree at 5b0d54f (`git archive`: `app/windows`, `brand/icons`, `packaging/windows`,
`proto-cuda/windows-app`, `proto-cuda/{host.cu,build.bat,README.md}`, `proto-opencl/{host.c,build.bat,README.md}`, `wsl2`, the two
worker `.rc` files), `--dir jobs --extract --extract-dir fb-0310 --fresh`.
2. `run` job `fb-installer-0310` to ae432dc7 (PowerShell, 20 min): `cargo --version` if any; the packaged json copied and extended; the
inputs zip downloaded and hash-checked; `BUILD-APP.bat`; `make-payload.sh` under Git Bash with `IGNEUM_APP_EXE` = the WSL engine
(`\\wsl$\Ubuntu-24.04\root\igneum-build\...\igneum-app.exe`, its sha256 checked against 4564f850...); `build-installer.ps1 -Payload <folder>
-Version 0.3.10` (with or without `-NoRcedit` per the word); `igneum-app.exe --version`, `Igneum Miner.exe --version`; `send.ps1` the
installer and the zip; every version and sha256 in the report.
3. On the Mac: the two files into the downloads folder, hashes equal to the report, the DLL gate, then
`node tools/ship-app.mjs 0.3.10 ... --from dmg` (preflight, then dmg already, copy, manifest, deploy, verify, console; the fetch step is
skipped by `--from`, the console item carries no run id), then the rollout as planned.
What the gate records in this plan if the fallback ships: the installer marked "PC-built on ae432dc7, GitHub run owed"; non-reproducible
(the runner's engine is the MSVC target, this one the GNU target from WSL; Inno Setup's version from winget; cl.exe 19.51.36260; Git Bash's
version; Windows 11 build 26200); the sha256 and size of the engine, the host, the payload zip and the installer; the GitHub run's id and its
own installer sha256 when it lands (they differ by construction); and the next cut goes back through the runner.
The coordinator asked at 21:3xZ whether the Mac mingw path was now faster and safer; the answer stands as below (there is no such path to an
installer), so PC 1's queue position was kept for attempt 4.
What does not exist: a Mac mingw build of `host.cpp` (the coordinator's "how 0.3.7 shipped"): 0.3.7's node exes were cross-built on the Mac
and its installer still came from the runner (release-0.3.6 plan, section 9); the host has only ever been built by `BUILD-APP.bat` with MSVC,
on the runner or on a PC. So if PC 1 lacked MSVC the next fallback would be new work, not a known path; PC 1 has MSVC, so it is not needed.
### 7d. The fallback, run (the coordinator's word at 21:00Z: GitHub's status page "Actions: major outage", the sixth queued run)
The PC 1 window: asked of the Counter ASIC coordinator (it schedules PC 1 tonight) at 21:0xZ; granted at 21:09:24Z when its reproducible
benchmark released the machine (every card restored; the RX 9070 XT back on the bus). `fb-tree-0310` (fetch) published 21:10:59Z, ran
21:11:32 to 21:11:33Z: the zip (709,814 bytes, sha256 ok) extracted into `%LOCALAPPDATA%\igneum\app\jobs\fb-tree-0310\fb-0310` (the
script's expected path was wrong and it now finds the tree by search). `fb-installer-pc1` (run, 25 min cap) published 21:19:28Z, FAILED at 21:20:05Z with exit 2 after 2 s: the script, not the build.
Its own lines: the tree found (82 files), `cargo on Windows: none` (so the engine is the WSL GNU-target build), Git Bash 5.3.15, Windows
10.0.26200, and then `engine not found at \\wsl$\Ubuntu-24.04\root\igneum-build\...\igneum-app.exe` with PowerShell's
`ItemExistsUnauthorizedAccessError`: the WSL tree belongs to root and the `\\wsl$` share refuses it to the app's non-elevated session. The
build jobs read that tree with `wsl -u root`, so the script now copies the engine out with
`wsl.exe -d Ubuntu-24.04 -u root -- bash -c "cp ... /mnt/c/.../fb-0310-work/igneum-app.exe"`. Republished as `fb-installer-pc1-2` at 21:2xZ
(the Counter ASIC coordinator kept PC 1 for it: "a 2-second exit 2 is the script, not the build", 5-minute reply window met).
`fb-installer-pc1-2` (21:23:08Z) failed at 21:23:58Z, exit 2 in 4 s: `/root/igneum-build/app/igneum-app/target/...` does not exist (the build
job keeps ONE target dir for the whole job; the acceptance script had read the node exes from `/root/igneum-build/target/...`): the script
now finds `igneum-app.exe` under `/root/igneum-build` with `find` and prints its sha256 from inside WSL. `fb-installer-pc1-3` (21:26:52Z)
failed at 21:27:20Z, exit 2 in 4 s: the inline `bash -c "..."` string lost a quote on its way through PowerShell (`unexpected EOF while
looking for matching quote`), the class the 0.3.6 cut closed for the app's own WSL calls (3811d8e, "every WSL script runs from a file") and
which this scratch script had reopened: the bash part is now written to `engine.sh` on the PC (LF, no BOM, the acceptance script's own
pattern) and run as `bash <file> <arg>`. Three 4-second failures of the script, none of the build; by the Counter ASIC coordinator's rule
its 10-minute measurement takes PC 1 first, then a read-only path probe (every path the script needs, found and printed), then attempt 4.
Before attempt 4 one more change, after reading the fixed block: `find ... | tail -1` would take the NEWEST `igneum-app.exe` under
`/root/igneum-build`, and other agents' build jobs have run on PC 1 since mine (job-console's, pack-loop's), so the sha check could stop
attempt 4 on another tree's engine. The engine this plan already holds and verified (4564f850..., 2,962,944 bytes, from PC 1's own job
`build-20261005-192450`) now travels INSIDE the tree zip (`fb-tree-0310b.zip`, with it under `app/igneum-app/target/x86_64-pc-windows-gnu/release/`),
and the script reads nothing from WSL at all; the path probe checks that file too.
Attempt 4 was never published: at 21:30:29Z GitHub's runner acquired try 6 (run 37374158235, dispatched 21:10:04Z on 5b0d54f) and it went
GREEN (the parse job 21:24:04 to 21:24:57Z; engine, window host, payload, installer, smoke run 21:25:18 to 21:30:29Z; the G13 inputs step
against the 21d4c73c inputs of 7a). The recipe's own installer therefore ships; the fallback stops here with nothing built on PC 1, PC 1
released to the Counter ASIC coordinator at 21:31Z, and the prepared pieces (the tree zip with the engine, the installer script, the two
probes, the runbook steps) kept in the scratchpad `r0310/fb/` as the rehearsed path for the next outage. Cost of the detour: three
4-second script failures on PC 1 and about 70 minutes of the shipper's attention; the gate record "PC-built, non-reproducible, GitHub run
owed" is not needed.
### 7e. The ship (21:31 to 21:40Z)
`OTA_SKIP=1 CONSOLE_SKIP=1 packaging/windows/fetch-ci-artifacts.sh 37374158235` (21:31:21Z): the installer and the payload zip into the
downloads folder; the DMG copied beside them. The payload zip holds the rebuilt workers (igneum-worker-cuda.exe 85cc357b..., 1,510,912;
igneum-worker-opencl.exe afa73a32..., 445,952: both differ from 0.3.9's f50e19f2.../1abc673e... and from the PCs' hot-fix pair), the PC-built
node 3adb01a7..., and the runner's MSVC engine 496c4883... (3,361,792). The first ship run (21:32:02Z) stopped in preflight: the tree was 4
commits behind origin/master (the coordinator's explorer merge, 9746391: docs, site, observer, ci.yml; no app, node or packaging file), so
`origin/master` was merged as ff873f6 (37 files, no conflict), pushed 21:32:27Z (the hook's site flip restored), and the state file kept
`sha` = 5b0d54f, the CI commit, as the 0.3.6 and 0.3.9 cuts did.
```
node tools/ship-app.mjs 0.3.10 --node vendor/igneum-node-0310 --branch release-0.3.10 --public \
--activation-height 135200 --deadline-note "finality v3" --notes "<the seven changelog lines; node 21d4c73c>" --from ci
```
| Step | Result |
|---|---|
| preflight | ok: tree ff873f6 clean, 0.3.10 in all 6, fork 21d4c73c, gh igneum-labs, live inputs 21d4c73c built 19:24:41Z |
| ci | already: run 37374158235 green |
| fetch, dmg, copy | already (above) |
| manifest | 0.3.10 mac+windows, signed (key 8f186e37...), verified locally; `consensus` carried over from the folder's 0.3.9 manifest: `activation_height` 135200, `deadline_note` "finality v3", `override` the four-field object; and in `dl/public/` (with the wallet 0.1.4 manifest) |
| deploy | one deploy, 21:33Z |
| verify | the token folder: `igneum-app-latest.json` 0.3.10, signature ok, mac 151687c5..., windows 24e58849...; the public folder: every file HEAD 200 with the local size (the DMG, the installer, the 0.3.9 HiveOS package, the wallet DMG, the four `/public/` aliases, the two manifests and their signatures, `igneum-downloads.json`), but the byte comparison of a json file against the edge still failed after 12 tries at 21:37Z and again on a resume from `verify`: the edge cache, the class of the 0.3.6 and 0.3.9 verifies (section 11 if it does not clear) |
| console | (pending: `--from console` after the verify clears) |
| File | sha256 | Size |
|---|---|---|
| Igneum-Miner-0.3.10.dmg | 151687c5672553c571af7224a8e4228348135b8a313fc89e6641db7ae21c85b3 | 41,383,375 |
| Igneum-Miner-Setup-0.3.10.exe | 24e58849f2b517e8c5579455e8c9d8376ff7dd27052f458ef91913cbc48abc88 | 49,858,273 |
| igneum-windows-app.zip | 16b68b7bf625a946ac62a22413982b38c6649a124d799b08424a070f1b34810e | 71,809,486 |
The installer is 30 MB larger than 0.3.9's (19,836,912): the two NVRTC DLLs (93 MB unpacked) ride with the rebuilt CUDA worker now.
`update-now-0310` to all, published 21:39:59Z (apps woken, stamp a23f594a). Timing with the Counter ASIC coordinator (it schedules PC 1 tonight): PC 1
was released by its hot-table job at 21:35:39Z, so one update-now went to every machine; its era measurement starts on PC 1's 0.3.10 STATUS line.
## 8. The machines after the publish (manifest live 21:33Z, update-now 21:39:59Z)
Baseline 21:40:31Z: Mac d937c69d app 0.3.9 (its card reads node 1, a24ab01a), PC 1 ae432dc7 0.3.9 node a24ab01a DAA 133,903 141.5 MH/s,
PC 2 1ccfe586 0.3.9 node a24ab01a DAA 133,945 0.0 MH/s (its 5090 worker off since a job's `/api/resume` at 21:25:11Z that the 0.3.9 app
answered and never acted on, section 11), Sam's Mac 3a9bf309 quit 53 min earlier, PC 37ba0461 0.3.9 node `2.1.0` 2.0 MH/s. A machine
counts as updated when its engine logs the 0.3.10 header, its node reports a DAA score and its miner a hash rate on 0.3.10. The two checks
asked by the coordinator: (1) the workers start without the seed-words error on the first try, on the rebuilt pair; (2) the Mac's card
shows node 1's digest, by design. C5: the provers' "stopped after" lines (shards aborted by the restart).
| Machine | On 0.3.10 | Its log |
|---|---|---|
| PC 1 ae432dc7 (Windows) | engine restart 21:40:41Z (run `win-ae432dc7-20261005-214041`), 42 s after the job; `[ok] updated to Igneum Miner 0.3.10 from 0.3.9` 21:40:42Z; `igneumd started` 21:40:46Z (the PC-built node from the new install path); `node proof verifier: command` and `reported: command` +6 s; `cards: NVIDIA GeForce RTX 5090 [discrete, off] \| AMD Radeon(TM) Graphics [integrated, off] \| AMD Radeon RX 9070 XT [discrete, off]` (hotplug's line: the iGPU integrated and off, the two discrete cards then started); miners started 21:40:47Z (nvidia-ae432dc7-1, the bundled `igneum-worker-cuda.exe`) and 21:40:48Z (amd-ae432dc7-3, the bundled OpenCL worker); `update to 0.3.10 complete` 21:42:12Z. Check 1 PASS: the NVIDIA miner's `epoch seed 66b26013... (daa 133967): CPU program and cache ready in 176 ms` 21:40:47Z, then `worker: info first pack packs\devnet: nvrtc 170 cache 4 dataset 23 check 249 race 37802 ms variant base; self-test PASS` and `worker: ready cuda NVIDIA_GeForce_RTX_5090 ... prepare 1 path nvrtc 12.8` at 21:41:25Z, no `epoch seed bytes do not give` line; STATUS 45.1 MH/s wall at 60 s (124.1 inside jobs), 124.3 MH/s inside jobs from 90 s on; the console 141.4 MH/s at 21:45:17Z (both cards). The node log: `igneumd/2.1.0` (no commit: the PC build, section 11), `Calibrated v1 fees ... 210000`, digest 1f4b44255fcd2ea8f75664ed47200f409186ddd2292960c9e2cf95bbbdc11505, peers node 1 (192.168.68.64) outbound and inbound and the seed 188.245.5.161, flows registered at THEIR protocol version 13 (a v14 node beside v13 peers, the mixed-fleet case of section 10, measured) | |
| Mac d937c69d | engine restart 21:40:51Z (run `mac-d937c69d-20261005-214051`), 52 s after the job; `[ok] updated to Igneum Miner 0.3.10 from 0.3.9` 21:40:52Z; `a node already answers on 127.0.0.1:26610; using it (it is not stopped by this app)`: the app attaches to node 1 as it has since 17:45Z, so its card shows node 1's version and digest (check 2, by design); `cards: Apple M5 Max [apple, off]` (its miner was off before the update too); `update to 0.3.10 complete` 21:42:22Z | |
| PC 2 1ccfe586 (Windows) | its 0.3.9 app fetched the woken jobs file at 21:40:33Z (`1 new for this machine, 1 queued`) and queued the update behind the aggregation-cost agent's running job `agg-cost-pc2-2` (one job at a time), so the update ran at 21:49:24Z when that job ended: installer downloaded and verified 21:49:27Z, `quit: stopping the miners, then the node` 21:49:29Z (no prover "stopped after" line: no shard in flight), engine restart 21:49:41Z (run `win-1ccfe586-20261005-214940`), 9 min 41 s after the job; `igneumd started` 21:49:42Z; `cards: NVIDIA GeForce RTX 5090 [discrete, off] \| AMD Radeon(TM) Graphics [integrated, off]`; miners started 21:49:44Z. Check 1 PASS: `epoch seed 66b26013... (daa 134395): CPU program and cache ready in 169 ms`, `worker: ready cuda NVIDIA_GeForce_RTX_5090 ... first pack ... self-test PASS` at 21:50:21Z, no seed-words line; STATUS 120.7 MH/s inside jobs at 60 s, 120.5 at 90 s; the console 123.2 MH/s at 21:53:04Z. This also ended the `/api/resume` no-op (its 5090 had been off since 21:25:11Z). The node log: `igneumd/2.1.0`, digest 1f4b4425..., flows at version 13 with node 1 and the seed (still a24ab01a for 10 s more) and at 14 with PC 1. The prover: `prover: host /opt/igneum/igneum-prove-host (WSL2), CUDA`, the pinned ids, then three assigned shards (22126, 51922, 84099) each failed with `Failed to create the CUDA prover impl: CudaClientError: Connect(Os { code: 13, kind: PermissionDenied })` at 21:49:56, 21:50:12 and 21:50:32Z: PC 2's prover is dark after the update (section 11) | |
| PC 37ba0461 (Windows, the US laptop) | its 0.3.9 app (run `win-37ba0461-20261005-202247`) downloaded and verified the installer at 21:40:52Z, started it at 21:40:53Z (`per-user install, no administrator prompt`), stopped its miners and node at 21:41:16Z, and no 0.3.10 engine run had reported by 21:56Z (the console: `STOPPED (update)` for 14 min): the install is in progress or stuck on the owner's machine; nothing to drive from here (section 11) | |
| Sam's Mac 3a9bf309 | quit since 20:47Z (0.3.9); takes 0.3.10 when it is started | |
C5, the shards aborted by the restart: PC 2's engine logged no prover "stopped after" line at its quit (21:49:29Z) and PC 1 runs no prover;
the Mac's prover was off. Observed count: 0. The coverage numbers measured across the restart carry no abort from it.
The ship's last step, `--from console` (21:55:08Z): item #364 "Igneum Miner 0.3.10 shipped (mac+windows)"; the tool's closing line
"manifest 0.3.10 published 2026-10-05T21:32:40Z".
## 8. The machines after the publish
(pending)
## 9. The hand nodes and the seed (21:49 to 21:50Z)
Moved while PC 2 and PC 37ba0461 were still on 0.3.9 (their updates gated by another agent's job and a slow download): the digest does
not change with 0.3.10 and a v14 node beside v13 peers is the measured mixed-fleet case (section 10), so moving them shortened the mixed
window.
| Node | Command | Result |
|---|---|---|
| The observer, then node 1 | `IGNEUMD=<fork>/target-integration/release/igneumd IGNEUMD_COMMIT=21d4c73c infra/devnet/restart-hand-nodes.sh '<the four-field object>'` (the branch's script: `grep -c` for the commit string, the object written to `/tmp/igneum-devnet/override-v3.json`, the previous file kept with a stamp) | observer restarted 21:49:38Z (pid 67770), node 1 21:49:50Z (pid 67963, caffeinate 67965); both print `Calibrated v1 fees ... from DAA score 210000` and digest 1f4b44255fcd2ea8f75664ed47200f409186ddd2292960c9e2cf95bbbdc11505; the Mac app's card follows node 1 from here (its node line reads 21d4c73c) |
| The seed 188.245.5.161 | `IGNEUMD_LINUX=infra/cross/out-0310/igneumd IGNEUMD_LINUX_SHA256=40be0e14... infra/devnet/restart-seed.sh '<the same object>'` (the zig cross-build of 21d4c73c, glibc 2.36 target; the script checks the sha on both sides, keeps the previous binary as `igneumd.prev-035` and the previous override file with a stamp) | restart 21:50:15Z, unit active, MainPID 125195, `igneumd/2.1.0-21d4c73c`, `Calibrated v1 fees ... 210000`, digest 1f4b4425... |
Peers after the restarts: the observer and node 1 connected to the seed within 20 s and register flows at protocol version 14 with each
other and the seed (node 1's inbound from the observer 21:50:08Z, node 1 to the seed 21:50:20Z, the observer to the seed 21:50:38Z; the
seed's three inbound peers from this Mac's address at 21:50:20, 21:50:27 and 21:50:38Z, all at 14). PC 1's node (started 21:40:46Z, before
the hand nodes moved) registered flows at version 13 with node 1 and the seed (then a24ab01a) and keeps them; PC 2's node (21:49:42Z) at 13
with node 1 and the seed (10 s before their restarts) and at 14 with PC 1: the mixed fleet of section 10, measured on the live network, with
no refusal and no drop.
### 9a. The digest sweep (21:50Z)
| Node | Binary | Digest | Read from |
|---|---|---|---|
| PC 1 app node ae432dc7 | the PC-built 3adb01a7... (`igneumd/2.1.0`, no commit string) | 1f4b44255fcd2ea8f75664ed47200f409186ddd2292960c9e2cf95bbbdc11505 | its node log through the log intake (run 214041) |
| PC 2 app node 1ccfe586 | the same | 1f4b4425... | its node log (run 214940) |
| The Mac app d937c69d | node 1's (attached; its card's commit string is the app's own reading from its start at 21:40:51Z, before node 1 restarted, and refreshes on the app's schedule; its status line reads node 1's height and peers) | node 1's | |
| Node 1 | 21d4c73c Mac arm64 4bb356f4... | 1f4b4425... | `/tmp/igneum-devnet/node1.out` |
| The observer | the same | 1f4b4425... | `/tmp/igneum-devnet/observer-v4.out` |
| The seed | 21d4c73c Linux 40be0e14... | 1f4b4425... | the journal through `restart-seed.sh` |
| PC 37ba0461 | (its update in progress) | (pending) | |
| Sam's Mac 3a9bf309 | quit since 20:47Z | (pending) | |
One digest, equal to the N3 publish's (`finality-v3-devnet-publish.md`) and to this cut's three readings on the fork (section 3): 0.3.10
moved no parameter, as measured.
## 10. The mixed fleet during the window
Between the manifest publish and the last app's update, old (a24ab01a) and new (21d4c73c) nodes share the devnet. What the two
agents' reports say about whether they can disagree:
| Change | Old and new nodes together | Source |
|---|---|---|
| D, transaction relay (PROTOCOL_VERSION 13 to 14) | They connect: a v14 node sends the three new messages only to peers at version 14 or later; a 13 peer never sees them (a node drops a connection on an unknown payload, which is why the version moved). Blocks, headers and the handshake are unchanged, the digest is unchanged. Only the mempools differ: a transaction sent to an old node is included only by that node's own templates, as before; a new node's pool converges with other new nodes'. No disagreement about the chain | the tx-gossip commit message e242acd0, the coordinator's line |
| C, the certificate-driven reorg | A consensus-behaviour change with no digest change: an old node keeps the shipped behaviour (a certificate over a block off its chain stays pending and it never reorgs to it), a new node locks that block and moves to the heaviest tip through it. The two CAN disagree on the selected tip after a partition heals while the fleet is mixed (exactly the C4 shape); the c4 agent's rollout note: it converges when every node is new. On the devnet tonight there is no partition in progress and one network, so the window carries no live split; the sweep in section 9 checks every node's DAA within a few blocks of the others | the c4 agent's rollout note (coordinator, 18:2xZ), the bench-log C4 rows |
| A, B, E, F | Windows link settings, a test switch, the app's card handling and job exit codes: nothing on the wire | |
| The override object | The same four fields on every node before and after; the manifest carries it verbatim (section 7), so no node's digest moves | sections 4 and 7 |
So the window is safe for ordering (no digest change, no protocol break) and the only behavioural difference needs a partition to show;
the hand nodes and the seed move last (section 9), after every app node is on 21d4c73c, so the fleet is fully new within minutes of the
publish.
## 11. Open after the cut
The next cut (0.3.11), decided by the coordinator on 5 October 2026 night:
| Branch | What | Why not 0.3.10 |
|---|---|---|
| fork `pack-loop` 05ef0fa3 (`vendor/igneum-node`) | `write_pack_checked`, `export-pack` exit 3, the force-prepare at the epoch boundary, the miner's exit 44 that unlocks the app's capped re-export | a node change after the node was frozen at 21d4c73c with its suites, harness runs and Windows acceptance done; the app side (af983a7) with the attempt-aware workers is the fix that matters tonight and it is in |
| `job-console` 13755b9, then 3562f26 and whatever follows | one hidden-console builder for every elevated launch, the spawn check in CI, PC 1 console watchers; then power control as a setting, default off (the project lead: "if we don't have to ask then don't ask") | arrived after the tree closed (both times); repoints elevated-exit's `include_str!` test at `platform.rs` at its own merge |
| `opencl-rdna4-telemetry` 7adcd4c (`igneum-wt-rdna4-telemetry`, not pushed: master + a08c371 as 38c9eec + 7adcd4c) | `igneum-gpu-telemetry.exe` (ADLX, SetupAPI bus, PDH fallback; amdgpu sysfs on Linux) built by `build-windows.sh`, shipped by `make-payload.sh` and `push-inputs.sh`; the engine runs it at `-l 5` and fills power, temperature, fan, memory clock and utilisation on the AMD card; 82 app tests pass. The 9070 XT measured on PC 1: 198.9 W, 64 C, 17.73 MH/s, 0.089 MH/W against the 5090's 0.398 MH/W (job `tele-measure-1`) | arrived after the tree closed; a new shipped exe and an engine source |
| `ember-tune` 54ff1bc (+38ef711, `igneum-wt-ember-tune`; its agent's message at 21:2xZ) | Ember Tune: `ember.rs` replaces the NVIDIA power-only run in `sweep.rs` and the AMD single-lever sweep of 720b369; sits on cherry-picks of 7adcd4c and 13755b9+3562f26, so those merge first; 93 app tests, `relay/test/ember.test.mjs` and `ui/tune-line.test.mjs` in ci.yml; design in `docs/plans/ember-tune.md` | arrived after the tree closed |
| the rest of `opencl-rdna4` a08c371 | the OpenCL worker's duplicate-platform fold, `--readback select`, `--memprobe`, `detect.rs parse_opencl_list`, the 9070 XT bench-log entry | conflicts with gpu-hotplug in `detect.rs` and `host.c`; only its EXPORT_LOCK hunk shipped (854f9a8) |
| Item | State |
|---|---|
| The per-job build-inputs zip (68f14b9, c4's tooling commit; the coordinator's check after two agents' PC jobs ran on another agent's sources through the shared `build-inputs.zip` tonight, jobs build-20261005-191656 and -192537). Checked on this tree's own jobs in the live jobs file rather than a dry run (a dry publish would leave a stray entry in the file the ship deploys): `build-20261005-182405` (PC 1) pins `params.zip_url` = `.../build-inputs-20261005182334-74461.zip`, `params.sha256` 628e6dae..., size 8,266,818; `build-20261005-183131` (PC 2) pins `build-inputs-20261005183051-83796.zip`, c83ebb2a..., 7,271,492; both shas equal the local zips of those names, and 15 per-job zips sit beside the folder default. So a job published through `tools/build-job.mjs` pins the zip it just pushed. The residual: `packaging/ota/publish-jobs.sh add --kind build` WITHOUT `--zip` still defaults to the shared `$DEST/build-inputs.zip` (line 307); nothing refuses that name. A hand-added build job can therefore still pin whatever the folder default holds | build-job.mjs path closed; the hand path is the first item of the next cut: refuse `--kind build` without `--zip`, or default to the newest per-job zip |
| `infra/cross/build-linux.sh` resolved a relative `TARGET_DIR` inside the node source after its `cd`, so the copy step shipped the previous build's bytes (twice tonight, caught by the commit string in `strings`). Fixed here (e0a5fd1). The class: any script that takes a directory argument and changes directory before using it; `proto-cuda/windows-node/cross-build.sh` takes the node worktree as `$1` and `CARGO_TARGET_DIR` from the environment (the 0.3.9 cut passed a relative one and it worked only because that script does not cd); a CI check for the shape is owed | fixed on the branch; the check owed |
| hotplug's `proto-opencl/host.c` change (the worker's `--list` prints the PCI address, the key to one row per physical card): no prebuilt OpenCL worker is in the payload inputs (the live zip holds igneumd.exe, igneum-miner.exe and three DLLs), the payload carries `host.c` and `build.bat` and the app builds the worker on the PC from them (`engine.rs build_worker_from_source`), so the change ships inside the 0.3.10 installer; whether an app that already has a worker exe rebuilds it from the newer source was not read here. Check on the console after the update: the PCs' card rows carry the PCI address (and one row per AMD card) only if the worker was rebuilt; else a rebuild is the hotplug agent's follow-up | open: told the hotplug agent |
| `kaspa-consensus` test `ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list` (c4-fix) failed once under the six-package parallel run on PC 2 with `UnexpectedDifficulty(487112096 vs 487129281)` in `mine_on_all` (section 3) and passes alone, twice. The helper builds a block on one `TestConsensus` and inserts it into others; the expected difficulty of the receiving node differs when the run is slow, which points at a wall-clock dependence (difficulty v2's sanitised clock per header). For the c4 agent: pin the clock or the timestamps in that helper, as the finality tests do for the cache queue. Until then a parallel six-package run can fail this test under load; the suites are run as consensus alone plus the other five | open |
| `site/build.mjs` is not idempotent: one bench entry's label alternates between "RTX 5090 first run" and "RTX 5090, memory-hard dataset" on every run of the same tree (three runs at 18:38Z: first-run, memory-hard, first-run), because the build reads its own `site/journey.json` back (line 248) and the label table at lines 203 and 204 matches against what the previous run wrote. Every push flips the two files through the pre-push hook, which is why the 0.3.9 and 0.3.10 cuts both met a modified tree after the push | open: build the journey from the sources only (never from the previous output), then have the hook refuse a push whose build differs from the committed site |
| The Windows payload carries the PC's three mingw DLLs (34 MB of inputs, `libstdc++-6.dll` alone 26.3 MB unstripped) although the exes import none of them since housekeeping A; the installer grew 5.2 MB | open: drop the DLLs from `push-inputs.sh` and `jobbuild.rs`'s copy once no shipped fork needs them, or strip them |
| The two worker exes' version blocks say 0.3.0 (`proto-cuda/nvrtc/igneum-worker-cuda.rc`, `igneum-worker-opencl.rc` are not among the six files `ship-app.mjs` bumps) | open: add the two `.rc` files to `VERSION_FILES` |
| The app's re-export of a refused pack (af983a7's `watchdog.rs` `PackRebuilds`, 3 per epoch) waits for the miner's exit 44, which the 21d4c73c miner never emits: a refusal in 0.3.10 shows the notice and restarts the worker, no re-export. The fork-side `pack-loop` 05ef0fa3 (the miner checks every pack, rebuilds a refused one, exit 44) is the other half | next node cut |
| C1 (the consequences reviewer, 20:0xZ): the 0.3.10 node's `igneum_exportSegments` (fork `igneum/exec/src/rpc.rs`) writes no `daaScore` and no `feesV1ActivationDaa` per segment, which the 0.3.9 exporter needs to replay both sides of the fee switch (`export/src/main.rs`: "a dump without `daaScore` is accepted only when the switch is never or 0"). Measured here: the handler (90 lines from `rpc.rs` 849 in `vendor/igneum-node-0310`) carries neither key (`daaScore` appears in the fork only in other RPCs, lines 239, 423, 819); the reviewer names `vendor/igneum-node-pv1` at eb32c645 (on 21d4c73c) as the carrier: its `rpc.rs` writes `daaScore` per segment (line 961) and `fees` and `feesV1ActivationDaa` at the top level (line 982); confirmed here at 20:4xZ (`git -C vendor/igneum-node-pv1 grep -n feesV1ActivationDaa -- igneum/exec/src/rpc.rs`: line 982); my first read of that path at 20:1xZ was wrong. So at H = 210,000 (about 19:50Z on 6 October) every 0.3.10 prover's export of a post-H block fails or meters with the wrong table and the fleet's provers go dark, unless the next node cut carries that RPC onto every prover before H, or H is republished later. The mechanism (the proving agent, 20:1xZ): the app's prover calls the exporter with no fee flags, so from H every app prover on 0.3.10 cuts with the prototype table and every statement is vetoed. The two closes: (a) the proving-v1 fork (eb32c645, on 21d4c73c) on every prover before H through 0.3.11; (b) republish H = tip + 86,400 by the fee-switch plan's rule. Decision due 16:00Z on 6 October; the coordinator, the proving agent and the 0.3.11 shipper hold the same line | open, dated: (a) or (b) by 16:00Z on 6 October |
| C5 (the same reviewer): the app kills its prover child on quit (`prover.rs` 266 to 272), so the `update-now` of this rollout aborts whichever shard each prover has in flight (up to 37 s each, no payout). Accepted as the cost of the restart; the count is read after the rollout from the provers' "stopped after" lines (section 8) so the coverage numbers measured across the restart are read with it | recorded at the rollout |
| `/api/resume` answered ok on the 0.3.9 app and never restarted the miners (PC 2's 5090 worker "off" with the card holding 1.7 GB from a job's resume at 21:25:11Z until its 0.3.10 restart, the iGPU miner too; the Counter ASIC coordinator, 21:4xZ). The class: a resume that reports success without a miner restart. The app should re-check the miner processes after a resume and report a failure | open: next cut |
| PC 2's prover after the 0.3.10 restart: every assigned shard failed at once with `CudaClientError: Connect(PermissionDenied)` (section 8) from 21:49:56Z. RESOLVED at 22:01Z by a socket fix on PC 2 (the Counter ASIC coordinator's agents; the SP1 CUDA prover's client socket permission), after which PC 2 proved and was paid every 40 s. The proving agent is adding a log line that names the cause on the app branch for the next cut | closed 22:01Z |
| PC 37ba0461 (the US laptop) started the 0.3.10 install at 21:40:53Z, stopped its miners and node at 21:41:16Z and had not come back by 21:56Z: the per-user installer on the owner's machine, nothing to drive remotely | open: the console shows when it returns; its 0.3.10 line and worker start are read then |
| `dl/public/igneum-downloads.json` (the unsigned index the site's download page reads) alternates at the edge between the new bytes and the previous ones for over 20 minutes after the deploy (one fetch byte-identical at 21:52Z, the next three not): different edge nodes behind one hostname. The two signed manifests were byte-identical and verified from the first check. The ship's verify step counts it as a failure and refuses to post the console item, so the item was posted with `--from console` | open: the verify should accept the index after the signed manifests pass, or retry it for longer; the site serves the previous version's buttons from a stale edge until it settles |
| The console's machine card keeps a machine's LAST non-empty node commit string: PC 1's card read `node 2.1.0-a24ab01a` for 17 minutes after its node had restarted as the PC build (`igneumd/2.1.0`, no commit), while PC 2's card read `2.1.0` at once; the Mac's card kept node 1's a24ab01a after node 1 moved to 21d4c73c. A card's commit string is therefore not a fact about the running node until the app re-reads it; the node log is | open: the card should show the string the app last READ, with its time, or nothing |
| The site build, run on a tree with conflict markers in `site/journey.json`, fails silently and leaves the markers (it reads that file back, the non-idempotence above): the first merge commit of this cut to master carried markers in two site files for one minute and was amended (the pre-push hook would have refused it, but the check must not depend on the hook) | fixed by hand here; the build should refuse a journey.json that does not parse and say so |
| The PC-built node binaries embed no commit (the build inputs zip has no git dir, `build-info` falls back to the bare version): the console shows PC 1 and PC 2 as node `2.1.0` with no commit after 0.3.10, as the 0.3.6 PC build did. The build inputs manifest records the fork commit (2f88a82f and later) | open: a commit stamp through the build job (`jobbuild.rs`) is an app change, not tonight |

View file

@ -0,0 +1,250 @@
# Igneum Miner 0.3.11: program class v3 (Counter ASIC 2.0) and proving v1 on the devnet, 5 October 2026
Release engineer, from 22:39 UTC, on the coordinator's instruction under the project lead's delegation ("Counter ASIC 2.0 fully deployed",
"deploy what is absolute best" for proving v1). Worktree `/Users/joshm/Projects/igneum-wt-ship0311`, branch `release-0.3.11`,
assembled by the Counter ASIC coordinator from master b38f3de (the 0.3.10 merge) and taken over at its merge tip b968ee0 so there
is one ship, not two. Fork worktree `vendor/igneum-node-0311` UNDER the release tree (the node links `../../../../igneum-pow`, the
release tree's crate), branch `release-0.3.11-node` at 89dfcb95 (on 21d4c73c = 0.3.10's node, with proving-v1 ece42979 and the
digest re-pin). The 0.3.10 recipe (`release-0.3.10.md`) throughout; every Mac build under the main checkout's lock; every PC
job and publish from this worktree's tools (the signed envelope, the per-job zip names). Times are UTC.
## 1. What 0.3.11 carries
| Change | Where | State |
|---|---|---|
| Program class v3 (Counter ASIC 2.0): the era draw, the cache growth rule, the mixer x8, behind `program_class_v3_activation_daa` (keys on the epoch: `program_class_v3_first_epoch`); the workers carry the class, era and attempt rules on the serve protocol and refuse a pack of the wrong class or era | main `ca2-v3` fa3c932 (code 49c7e78; `igneum-pow`, the workers, the fast-time scripts, the plans); fork `ca2-v3-node` 89dfcb95 | merged (c9b0b2c) |
| Counter ASIC 2.0 docs, spec, site, evidence, the rollout plan and its gates (G1, G2, G3, G4, G4b, G6 green; G5 = the one-commit workers of this cut) | main `ca2-coord` 076c0ab then 57844e9 (C34) | merged (18605f4, 5cedcd4) |
| Proving v1 (spec 7.8): the aggregated segment record, the chain rule and the unproven rule behind `proving_v1_activation_daa`, with `proving_v1_segment_blocks` 8, `proving_v1_unproven_daa` 600, `proving_v1_aggregator_share_bps` 1000; the app's prover loop (the CPU path refused under 32 GB with the reason, the root-socket cleanup, the PermissionDenied line naming the cause); the resume fix (every stopped card re-armed, its pack re-exported, checked 90 s later); every worker gets `--prepare-packs` in the platform's path form (the Mac's Metal worker takes a class v3 day from the prepared pack) | main `proving-v1` 22c2363 (its agent: the final code tip; c36dfea after it is docs only and waits); fork proving-v1 ece42979 inside 89dfcb95 | merged (5cbb796) |
| CI: `bash-body-check.sh` (inline bash bodies in PowerShell job scripts parse), `kit-path-check.sh` (a run job tests its fetched kit before use, C32), `prover-socket-check.sh` (every root prover playbook unlinks the GPU server's socket) | main `bash-body-check` e3bd761 | merged (fe1ecdb; ci.yml keeps the signer-pipe step, both new steps and proving-v1's socket step) |
| The consequences ledger, the proving-methods analysis, the ASIC-resistance history | `consequences` 99fd988, `proving-methods` e7e0db7, `asic-history` 9e4af7f | merged (5dffb1b, 0f5bfc3, b968ee0) |
| The pinned proving guests | unchanged (no prover drain) | |
Changelog line (the coordinator's words): "Igneum Miner 0.3.11: program class v3 (the era draw, the cache growth rule, the mixer x8) from epoch
N4/3600 and proving v1 from DAA N5; the resume fix; the Metal worker takes v3 from a prepared pack".
## 2. The branch
| Commit | What |
|---|---|
| b968ee0 | the Counter ASIC coordinator's merge tip (above), taken over at 22:40Z; its checks on the Mac: igneum-pow 53 + 4 + 19 + 7, the app 113 + 27 + 8, 0 failed |
| 21173c4 | `Igneum Miner 0.3.11: the six version files` (`--check`: 0.3.11 in all 6) |
| cc72f4a | CI on the merged tree, three findings fixed: the identity check's hostname pattern `MacBook` matched prose in `card-lifetime-2026-10-05.md` and `proving-methods.md` (reworded "Apple laptop"); the kit-path check flagged `tools/proving-v1/pc2-{memory-miner-on,memory-sweep,sp-curve}.ps1` for a bare `jobs\` literal in `WslPath (Join-Path ...)` (now `Test-Path` on the kit root, then `WslPath $kitFile`); the socket check flagged `tools/amd-prove/pc1-cpu-prove.ps1` and its `-sp` sibling, which the addendum said were allow-listed and were not (allowed, the CPU path starts no GPU server). The other agent's resolution had kept both ci.yml sides and bash-body-check's 24-line socket check; one slip of mine (a `git show :3:` redirect after that resolution had already committed) truncated that script to 0 lines in the working tree for a minute and was restored from HEAD |
| 23bc2b2 | `packaging/mac/packaged-config.sh`: the nine-field object (section 4) in the packaged line, `--test` passes (C34: a fresh install must start on the fleet's digest) |
| a94416e | the plan draft committed to the tree before the push (the reviewer's C37) |
| after a94416e | C38 (the reviewer through the Counter ASIC coordinator): the evidence rows, the litepaper's chip bullet, `chip-model-v3.md` and the rollout plan cite files on four Counter ASIC branches that never reached the tree. Taken as docs plus standalone sources under one gate: the diff against 23bc2b2 over every input a built artefact reads (`igneum-pow/src`, `app/`, `proto-cuda/nvrtc/{packfile.h,worker.cpp,cuda_api.h,build-windows.sh}`, `proto-cuda/{host.cu,build.bat,windows-app}`, `proto-opencl/{host.c,cl_dynamic.h,build.*}`, `proto-metal`, `packaging`, `proving`, `vendor`, `.github`) must stay empty apart from files no script compiles or copies. Merged: `ca2-analysis` ee42d7c (`sram-mirror.md`, `int8-matrix-family.md`, the dot4 probe sources: `proto-metal/dot4-probe.swift` is standalone, the DMG script compiles `main.swift` alone), `ca2-epoch` e95e8b5 (`epoch-length.md`), `prover-floor` cfe3d80 (`prover-floor.md`, `tools/prover-floor/` scripts, a playbook, `proving/prover-floor/sp1-gpu-6.8.1-floor.patch`, which nothing reads: the next cut's packaging row); `docs/bench-log.md` conflicted each time (append-only: both sides kept). `ca2-soundness` a465881 conflicted in `igneum-pow/tests/scratch.rs` (add/add) and `proto-metal/packbench.swift`, code the mixer merge already carries, so its merge was aborted and `docs/analysis/scratch-soundness.md` taken alone. So G5 holds: the workers, the Metal worker and the DMG built at 23bc2b2 correspond to the tip's built inputs |
Every check at cc72f4a: identity 0 hits over 220 files, copied-sources, pinned-guests, signer-pipe, bash-body 15 bodies in 28 files, kit-path 14 of
14 kits checked, prover-socket, no-conflict-markers, workflow shell 0 findings over 35 .ps1, relay tests 17, UI tests 23.
## 3. Builds and tests (the 0.3.10 recipe)
| What | Command | Result |
|---|---|---|
| The fork's Mac node, 89dfcb95 | `CARGO_TARGET_DIR=vendor/igneum-node/target-0311 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` from `vendor/igneum-node-0311` (under the release tree), under the lock; the target dir cloned by APFS from `target-0310` | 22:46:16 to 22:49:5xZ (3 min 19 s, incremental): igneumd bd7f043c453f4b3e9575e678912b71115227b09789d5ec0f367b8278f031d043 (41,386,016, `89dfcb95` in its strings); copied into the fork worktree's `target-integration/release/` |
| The seed's Linux node (glibc 2.36 target, zig) | `NODE_SRC=<abs fork> TARGET_DIR=<abs> OUT_DIR=<abs> infra/cross/build-linux.sh` (the 0.3.10 fix: absolute paths; `cargo clean -p kaspa-build-info` first), under the lock | 22:46:35 to 22:49:59Z (3 min 20 s): igneumd 63cf490d42483d3aa4e525eba9b4e4b68cae9090c32f409e745858defc381c52 (47,913,832, `GLIBC_2.34` at most, `89dfcb95` in its strings), igneum-miner a136d622... (9,861,280) |
| The two Windows workers (G5: one commit) | `proto-cuda/nvrtc/build-windows.sh` under the lock, tree 23bc2b2 | 22:46:43 to 22:46:50Z: igneum-worker-cuda.exe 2b3b8c92885442179f6bf2907c6f3eb453dc4a19908d90fd05981a09b7c2674c (1,536,512), igneum-worker-opencl.exe edc4a75da3b93d814caa69fd635010780d63d5b622ec24c3741d433c584f91e3 (478,208); both carry the resource block; both differ from 0.3.10's pair (85cc357b..., afa73a32...): the class, era and attempt rules are in them |
| The app | `cargo build --release` in `app/igneum-app` (target cloned from the 0.3.10 worktree), then `cargo test --release -p igneum-app`, under the lock | 22:47Z: `igneum-app 0.3.11`; tests ok 113 (lib) + 27 (ota-sign) + 8 (prove-verify), 0 failed |
| `igneum-pow` | `cargo test --release` in `igneum-pow`, under the lock | 22:47:33Z: ok 53 + 4 + 19 + 7, 0 failed (the class v3 vectors, the era draw, the mixer x8, the scratch soundness) |
| The prover host and export (the pin unchanged) | `cargo build --release -p igneum-prove-export -p igneum-prove-host` in `proving/igneum-prove` (the worktree needed the vendor links: `vendor/igneum-node-exec` and 45 others symlinked to the main checkout's, beside the real `igneum-node-0311` worktree) | 22:48Z: `--mode id` shard `0x2b1a81cb...`, aggregator `0x474678f3...` (the 0.3.9 pin: no prover drain); the nine real fixtures `--mode native` all ok |
| PC 1 build job (the node and the app, Linux and Windows) | `IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release node tools/build-job.mjs run --node vendor/igneum-node-0311 --target ae432dc7 --targets linux,windows --no-tests` from this worktree, its own zip `build-inputs-20261005224625-29789.zip` | job `build-20261005-224654`, published 22:46:54Z (PC 1 given by the Counter ASIC coordinator at 22:4xZ: Ember's collect job closed, the AMD sweep off tonight). (pending) |
| PC 2 combined job | `IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release node tools/build-job.mjs run --node vendor/igneum-node-0311 --target 1ccfe586 --targets linux,windows --node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows" --app-tests igneum-app` from this worktree (its own zip `build-inputs-20261005230716-50058.zip`); PC 1's job removed from the jobs file so nothing double-places the exes | job `build-20261005-230745`, published 23:07:45Z (PC 2 woken), the first job on PC 2's re-set schedule; started 23:09Z, done 23:16:49Z (469 s): the Linux stage, the Windows stage 264 s, the test stage `RESULT test node [kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows] exit 0 54 s` and `RESULT test app/igneum-app [igneum-app] exit 0 6 s`; 9 outputs verified and placed: igneumd.exe be8e83c07aeae5eb6842768071735289f3ef149ed9a7088592d8c54a4c252c08 (51,321,856), igneum-miner.exe 1ba1a249e2a21d52087a81f61f37cd2e1e3273c7ec8809ef9cfaa98f015895ce (10,987,520), igneum-app.exe ab104cc0... (3,065,344, the PC's; the installer's engine is the runner's), Linux igneumd d7a2715e... (49,193,960, glibc 2.39, HiveOS), igneum-miner 09d05ff6..., igneum-app 222f30c0... |
| PC 2 suites | the combined job above (the PC 1 job of the row above never started: PC 1's app is down, section 3a) | six node suites exit 0 in 54 s, the app suite exit 0 in 6 s (23:16:49Z) |
| The Linux workers for HiveOS | `infra/cross/build-workers-linux.sh` (zig, glibc 2.36 target) under the lock, 23:10Z | igneum-worker-cuda 4aaff27fcb26bc5fe98d2b311b9f95f099c59414a556af32080b82b41e8360db (6,755,568), igneum-worker-opencl 82d90890be36f9b795b218794062e388bfd9c6f7524fe671074cdec84c906259 (306,000), ELF x86-64 dynamic; untested on a GPU host, as the script says |
| The HiveOS package | `NODE_OUT=<the zig node> WORKERS_OUT=<those workers> VERSION=0.3.11 packaging/hive/make-hive-package.sh`, 23:11:30Z, then `packaging/ota/publish-public.sh --hive` into `dl/public/` (the ship's deploy carries it) | `igneum-hive-0.3.11.tar.gz` c606a17043c013c411086b4abd0f38a227aa8a7db9d5934157760a3da5a5edc9 (24,496,653); no override inside (section 5) |
| The DMG | `NODE=<fork>/target-integration/release/igneumd MINER=... packaging/mac/build-dmg.sh` under the lock, 22:50:0x to 22:50:51Z | `Igneum-Miner-0.3.11.dmg` b7e81d4f6f3af9f9179e29faa72c844b795df757dfb2e4cd1d56cd454e78e1e7 (41,592,041): engine 0.3.11, node 89dfcb95 Mac arm64, the 0.3.9 prover host and export, `igneum-bench` (the Metal worker) rebuilt from this tree (667,145 to 555,808 bytes, signed ad hoc), fingerprints 477bb0ef and ed9c4d2e; read back from the mounted image: `Contents/Resources/igneum-app.json` carries the nine-field `node_override_params` with N4 = N5 = 154,800 (C34) |
### 3a. PC 1's app down, and the relay task that hit the wrong machine (22:31 to 23:05Z)
PC 1's installed 0.3.10 app quit at 22:31:06Z (its last upload 22:31:08Z, run `win-ae432dc7-20261005-214041`; Ember's tune job on it
reported "aborted (the app is quitting)"; the quit's sender is C35 for the consequences reviewer) and did not come back, so PC 1 mined
nothing, its build job `build-20261005-224654` could not start and no update-now could reach it. Three relay tasks (#241 at 22:52:31Z, #243 at
22:55:10Z, #245 at 23:03:18Z) went to the relay machine named "PC1" to relaunch the app, the second ending an engine that answered nothing on
`/api/state` and the third ending every igneum process before the launch, as the app's own updater does.
They hit the wrong machine. The relay's "PC1" is the 1ccfe586 box, the console's PC 2: both PCs carry the hostname DESKTOP-KMCV30N, the only
relay agent runs on the 1ccfe586 box and was named PC1 when the clients were set up, and the relay's PC2 entry reads "never seen". The
intake proves it: PC 2 got two new engine runs, `win-1ccfe586-20261005-225528` and `-230330`, at the exact times of #243 and #245, while PC 1
has no run after 21:40:41Z; #243's "hung" engine, pid 26696 from 21:49:40Z, was PC 2's healthy 0.3.10 engine (my probe's "no answer" on
`/api/state` was its own fault, no token), and the node, three miners and three workers it found were PC 2's own (a 5090 and the iGPU; PC 1
would have shown the 9070 XT too). So PC 2, the box the night's measurements run on, was force-restarted at 22:55:28Z and 23:03:30Z, which
killed the aggregation-cost agent's job 3 (re-run owed, 20 min) and ended whatever followed; its app came back each time (after #245: pid 30484,
responding, node, three miners and both workers up, the card climbing at 23:04Z). PC 1 is exactly as it was: engine down since 22:31:06Z,
no relay agent on that box, unreachable tonight; it waits for the project lead in the morning and takes 0.3.11 through the manifest at its relaunch; the
fleet runs short its 141 MH/s until then. Told the Counter ASIC coordinator at 23:05Z; it re-set PC 2's schedule (this cut's combined build
and suite job first, then the prover-floor pair, the aggregation-cost re-run, "PC 2 clear", the M16 job) and ruled that no relay task goes to
"PC1" from anyone without its word. Every relay "PC1" reading tonight was PC 2 (the AMD agent's "9070 XT absent" probes read a box that has
no 9070 XT; the hardware events are being corrected). The relay machine should be renamed PC2 (`node tools/relay.mjs name <hostname> <name>`)
and the PC 1 box get its own agent (section 11). The Mac cross-build fallback for the Windows exes was announced and not started: the
combined PC 2 job replaced it within the minute.
C35's cause (Ember Tune, 00:2xZ): the second engine its measurement job starts on PC 1 reports 0.3.9 (the `ember-tune` branch's Cargo
version), sat under the manifest's `min_supported_version`, took 0.3.10 as urgent (over `auto_update = false` in its copied settings), ran
`ota-apply.ps1` at 22:31:05Z, and the per-user installer's PrepareToInstall quit the INSTALLED app at 22:31:06Z, which then hung on the pipe
its orphaned grandchildren held. For this plan's bookkeeping: PC 1's 0.3.10 came through my update-now at 21:40:41Z (`release-0.3.10.md`
section 8); the 22:31:05Z run was a second install of 0.3.10 over 0.3.10 by that engine, outside the rollout order, and it is what took
PC 1 down. The guard is `ember-tune` e600e63 (section 10).
## 4. The override object, N4 and N5, the digests
The object every node runs after the publish (the four live fields plus the five new ones):
```
{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000}
```
N4 and N5 are fixed BEFORE the DMG and the installer are built, because the packaged line (C34) must equal the manifest's object: DAA 136,967 at
22:45Z (PC 1's card) at 0.965 blocks/s puts the publish (about 23:40Z) near 140,200; tip + 14,400 is near 154,600; the first multiple of 3,600 at or
above it is 154,800 (N4's rule), which N5 takes too. At the publish the floor N - DAA >= 10,800 is checked; it holds until DAA 144,000 (about 00:45Z);
past that the line is re-pinned and the DMG and installer rebuilt.
The two digest readings on the 0.3.11 Mac node (bd7f043c..., ports 60975/60976, 22 s each, under `run`):
| Override file | Lines | Digest |
|---|---|---|
| none (the rolling-upgrade value: both new activations at never) | `igneumd/2.1.0-89dfcb95` | c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c, EQUAL to the node agent's pinned test on 79bd8e10 (22:49:59Z) |
| the nine-field object above | `Program class v3 from the override file: active from epoch 43 (DAA score 154800 rounded up to the epoch boundary at 154800, epochs of 3600 DAA)`, `Proving v1 from the override file: segment records paid from DAA score 154800, 8 blocks a segment, unproven after 600 DAA, aggregator share 1000 bps`, `Calibrated v1 fees ... 210000`, `Finality rule v3 ... 135200` | **0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888** (22:50:23Z): the value every node must print after the publish; the sweep in section 9 reads it on every node |
| the fleet's live four-field file (what every node runs today) | `Calibrated v1 fees ... 210000`, `Finality rule v3 ... 135200`, no class v3 or proving v1 line (both at never) | **4d8f8bb668828a3dcf7b783b995f3d3ebfde32a092dd1dbd5bf4373c5c65a62c** (23:13:54Z): the 0.3.11 BINARY alone flips the digest (the 0.3.10 node prints 1f4b4425... with the same file, `release-0.3.10.md` 9a): the new fields enter the digest even at never. This is the digest of step 1 (the reviewer's C40, the Counter ASIC rollout plan's section 4 step 1); both had assumed c562d70e..., which is the no-file case |
Three digests on the 0.3.11 binary, so two sweeps: step 1 moves every node to the binary (4d8f8bb6...) and step 2 moves every node to the
nine-field object (0139ab9d...). A node on either side of a sweep is refused by the other side (the handshake), so each sweep is one
window, as the fee switch's 8 min 47 s was.
## 5. The rollout order: two publishes, two sweeps (the reviewer's C39 and C40, the Counter ASIC coordinator's rule, the fee-switch shape)
Why two: the app writes the manifest's `consensus.override` at the manifest TAKE (`ota.rs` 661, `write_override`) and restarts its node with
it at the next safe window, whatever binary is installed; a 0.3.10 `igneumd` refuses a file with `program_class_v3_activation_daa` or the
proving v1 fields (`OverrideParams` is `deny_unknown_fields`) and dies at start, and the update then waits for a synced node (`engine.rs`
2211) until the slot minute or the 1,800-block rule forces it. One manifest with 0.3.11 AND the nine fields would take every 0.3.10 node
down at its next safe window: the 0.3.5 class the fee-switch plan named. And the 0.3.11 binary alone flips the digest (section 4), so the
binary move is itself a sweep.
| Step | What | Digest after |
|---|---|---|
| 1a | the observer, node 1 and the seed on the 0.3.11 binaries with the four-field file: `IGNEUMD=<fork>/target-integration/release/igneumd IGNEUMD_COMMIT=89dfcb95 infra/devnet/restart-hand-nodes.sh '<the four-field object>'`, then `IGNEUMD_LINUX=<the zig build> IGNEUMD_LINUX_SHA256=63cf490d... infra/devnet/restart-seed.sh '<the same>'`; the apps still on 0.3.10 are refused by them from this moment until each updates | 4d8f8bb6... on the three |
| 1b | the manifest: 0.3.11 with `consensus` carried over UNCHANGED (`--activation-height 135200 --deadline-note "finality v3"`, the four-field object, exactly as 0.3.10 shipped), `--public` (the HiveOS package rides along) | |
| 1c | update-now: the Mac (its card follows node 1) and the laptop first; PC 2 only on the Counter ASIC coordinator's "PC 2 clear"; PC 1 is down and unreachable (section 3a) and takes 0.3.11 through the manifest at its morning relaunch, refused until then. The watch: every app logs 0.3.11 and its node a DAA score at 4d8f8bb6...; every worker starts clean on the first try with the class-aware pair; the Mac's Metal worker takes the class v3 day from the prepared pack; C32: the agents whose kits sit on PC 2 republish their fetches after its update | 4d8f8bb6... on every reporting node |
| 2a | the floor: 154,800 minus the tip's DAA at least 10,800 (holds until DAA 144,000, about 00:45Z); past it N4 = N5 re-pinned to the first multiple of 3,600 at or above tip + 14,400, the packaged line, the DMG and the installer rebuilt; the DAA read sent to the Counter ASIC coordinator before 2b | |
| 2b | the hand nodes' and the seed's files switched to the nine-field object and restarted (the same two scripts), the manifest republished with `--override '<the nine-field object>' --activation-height 154800 --deadline-note "program class v3 + proving v1"`, update-now (the same order; PC 2 on "clear" again), the sweep | 0139ab9d... on every node |
| 3 | the plan's final sections, the merge to master (the live observer must not read stale: `public-api-check`'s other arm), the push, the report with per-machine times | |
The HiveOS package carries NO override: `packaging/hive/h-run.sh` line 31 starts the rig's node with `--devnet --appdir --rpclisten --listen` and
the peers, no `--override-params-file`, and no HiveOS package has ever carried one, so a rig's bundled node runs on genesis params and is refused
by every devnet peer (the HiveOS path is untested on a GPU host since 4 October). "Republish with the new override" therefore needs an
`h-run.sh` change (the file written from the Flight Sheet's extra config, as `PEERS=` is), which is the next cut's; tonight's package carries
the class-aware binaries only (section 11).
## 6. The push and CI
`git push -u origin release-0.3.11` at 3b0262f (23:18:29Z, the credential helper; the pre-push hook's site flip restored), `gh workflow run
windows.yml --ref release-0.3.11` -> run 37387737179, acquired at once (GitHub operational again), green 23:23:11Z (the parse job 23:18:39 to
23:19:25Z; engine, window host, payload, installer, smoke run 23:19:31 to 23:23:11Z, the G13 step against the 89dfcb95 inputs). The main
`ci.yml` run on 3b0262f (37387751432) failed in one step, the igneum-census release build (the class v3 fields missing from the census's own
initialisers, `fetch`'s new Layout argument, no Scratch/Hot match arms; pow tests, simulators and site jobs passed); the coordinator fixed it
in this worktree as 2a62735 (`igneum-census/src/main.rs` only; `git diff --stat 3b0262f 2a62735 -- app packaging igneum-pow proto-cuda
proto-opencl proto-metal` is empty, so the Windows artefacts of 37387737179 stand) and its run 37388453875 is green (pow tests and census
build, simulators, site). The 0.3.11 CI verdict is therefore run 37388453875 on 2a62735; the Windows build is run 37387737179 on 3b0262f,
the same app sources; the merge to master goes from 2a62735.
| File | sha256 | Size |
|---|---|---|
| Igneum-Miner-0.3.11.dmg | b7e81d4f6f3af9f9179e29faa72c844b795df757dfb2e4cd1d56cd454e78e1e7 | 41,592,041 |
| Igneum-Miner-Setup-0.3.11.exe | 84a21443f78598597f33cef307fa162e53c680dd90d29f02ceaf79ac5e231ee4 | 50,065,009 |
| igneum-windows-app.zip | ed0cf2a75a1de8897de33ddcdcdb4ea844eca6b38627d6b91c562700c9fbe5eb | 72,070,333 |
| igneum-hive-0.3.11.tar.gz (dl/public) | c606a17043c013c411086b4abd0f38a227aa8a7db9d5934157760a3da5a5edc9 | 24,496,653 |
## 7. The rollout, step 1 (the binary sweep to 4d8f8bb6...)
Baseline 23:19:40Z: tip DAA 139,642; the observer and node 1 on 21d4c73c at 1f4b4425...; the Mac app 0.3.10 (attached to node 1); PC 2
0.3.10 at 120.8 MH/s; PC 37ba0461 back on 0.3.10 at 2.3 MH/s; PC 1 down since 22:31:06Z (section 3a); Sam's Mac quit since 20:47Z.
| Step | Time | Result |
|---|---|---|
| 1a the observer | 23:23:54Z (pid 61754) | `igneumd/2.1.0-89dfcb95`, the four-field file, digest 4d8f8bb668828a3dcf7b783b995f3d3ebfde32a092dd1dbd5bf4373c5c65a62c |
| 1a node 1 | 23:24:06Z (pid 61864, caffeinate 61866) | the same |
| 1a the seed | 23:24:25Z (MainPID 125853) | `igneumd/2.1.0-89dfcb95`, the same lines; the a24ab01a... no: the 21d4c73c binary kept as `igneumd.prev-035` (the script's name) |
| 1b the ship (`--from ci`) | 23:24:44Z | preflight ok (tree 2a62735 clean, 0.3.11 in all 6, fork 89dfcb95, gh igneum-labs, live inputs 89dfcb95 built 23:17:51Z); ci, fetch, dmg, copy already; `consensus.override: carried over from the current manifest` (the four-field object), `activation_height` 135200, `deadline_note` "finality v3"; manifest 0.3.11 mac+windows signed (key 8f186e37...) and in `dl/public/`; one deploy; verify: the token folder's manifest 0.3.11, signature ok, mac b7e81d4f..., windows 84a21443...; the public folder's `igneum-downloads.json` not yet the local bytes at the edge (the 0.3.10 class), resumed from `verify` for the console item |
| 1c update-now, the Mac and the laptop | 23:28:06Z (`update-now-0311-d937c69d-37ba0461`, apps woken) | the Mac: the job ran 23:28:48Z, `Igneum Miner 0.3.11 is available: downloading (41 MB)` 23:28:49Z, engine restart 23:29:05Z (run `mac-d937c69d-20261005-232905`), 59 s after the job; `[ok] updated to Igneum Miner 0.3.11 from 0.3.10`; `a node already answers on 127.0.0.1:26610; using it`: the app attaches to node 1 (89dfcb95, 4d8f8bb6...), so its card follows node 1; `cards: Apple M5 Max [apple, off]`: its Metal miner was off before and after, so the prepared-pack check has no subject on the Mac tonight. The laptop (PC 37ba0461): (pending) |
| the console | 23:30Z | item #366 "Igneum Miner 0.3.11 shipped (mac+windows)" (`--from console` after the public index settled at the edge; the ship's own verify had refused it as 0.3.10's did) |
| the old side during the window | from 23:24Z | PC 2 and the laptop at 0 peers (refused by the new side) until each updates. The new side (node 1, the observer, the seed, the Mac app attached to node 1 with its miner off) has NO miner on it, so its chain STALLED at DAA 139,751 from about 23:29Z (the observer's `/api/stats` at 23:31:42Z: 139,751, age 1.8 s) until a miner joins it; the old side's fork grows only while a miner mines there, and PC 2's 0.0 MH/s from 23:29Z is the prover-floor sweep stopping its miners for its run (`floor-sweep-2`, started 23:21:17Z), not the refusal. Put to the Counter ASIC coordinator at 23:31Z with the numbers; its call: "PC 2 clear" the minute the sweep closes (about 23:36Z) and at 23:45Z at the latest, the restore job after the update, because an app restart under the sweep kills its job tree and its restore never runs; the laptop's 2 MH/s restarts the new side's chain when its install ends. The lesson for the next cut's plan: step 1a (the hand nodes and the seed first) moves the hub to a side with no hash until the first miner updates; the first update-now should go to a miner within the same minute |
| the Mac's miner (C42) | 23:39:10Z | the new side had NO miner: the Mac app's mining was a persisted `settings.paused` (cleared only by Resume; its STATUS lines read "0.00 MH/s, paused" since the update), node 1 runs no miner, the fleet's hash was on the refused old side. `POST <app.url>/api/resume` (the token path) answered ok at 23:39:10Z; the Metal worker compiled the program inline at 23:39:11Z ("not prepared": the prepared-pack path did not engage on this start, a G4b finding, section 11), raced 14 variants, 3.15 MH/s wall at 23:39:48Z; the new side's first blocks accepted 23:39:49, 23:39:51, 23:40:00Z. The stall: about 23:29Z to 23:39:49Z, 11 minutes of stopped DAA clock on the side every node ends up on. New check for step 1 of every digest-flipping cut: the new side has at least one miner before the first update-now |
| 1c update-now, PC 2 | 23:39:44Z (`update-now-0311-1ccfe586`, on the Counter ASIC coordinator's "PC 2 clear": `floor-sweep-2`'s first point had hung 18 min, nothing in flight to protect; its restore-and-diagnose job is the first PC 2 job after the update). PC 2 fetched the woken file at 23:40:13Z and logged `1 new for this machine, 1 queued`: update-now is itself a job and the app runs jobs one after another, so it waits behind the hung `floor-sweep-2` (started 23:21:17Z, cap 30 min, freed about 23:51:17Z); PC 2 reads `0.00 MH/s, waiting | node 139753 blocks, 0 peers, syncing` every 30 s meanwhile (refused by the new side, its miners stopped by the sweep). The class, for the next cut's plan: an update-now cannot pre-empt a running job; a hung job's cap sets the update's time | the queued job ran at 23:51:19Z, two seconds after the sweep's cap; installer verified 23:51:22Z; `quit: stopping the miners, then the node` 23:51:24Z; engine restart 23:51:30Z (run `win-1ccfe586-20261005-235130`), 11 min 46 s after the job and 3 min 35 s after PC 2 had received it; `igneumd started` 23:51:32Z; the prover's host and the pinned ids as before; `miner nvidia-1ccfe586-1 started` 23:51:43Z. Check 1 PASS at the first start: `epoch seed f4d9d3d8... (daa 140361): CPU program and cache ready in 166 ms`, `worker: ready cuda NVIDIA_GeForce_RTX_5090 ... first pack ... self-test PASS` at 23:52:21Z, no refusal. Then the hourly epoch boundary fell at DAA 140,401 (`SEED CHANGE ... f4d9d3d8... -> cb5b51cc...`, 23:52:21Z, 40 s after the first pack): the worker refused the new epoch's jobs (`epoch seed mismatch: this worker holds epoch f4d9d3d8..., the job is for epoch cb5b51cc...`, 80 lines over 80 s) while the app's attempt-aware path prepared the new epoch (`program pack checked for epoch seed cb5b51cc... day 20732: attempt 0`, the line the pack-loop fix added; `PREPARE sent ... class v2 (3559 DAA blocks before the boundary at 144000)`), the worker switched at 23:53:42Z, STATUS 113.4 MH/s wall at 23:54:14Z, the card 112.4 MH/s at 23:56:04Z. The Mac's worker crossed the same boundary in 241 ms (`prepared cb5b51cc... class v2`). PC 2's node joined the new side at once (its DAA tracks the observer's, 1 peer: node 1 at 192.168.68.64). The sweep's restore-and-diagnose job is the Counter ASIC coordinator's, after this |
| the laptop (PC 37ba0461) | silent since 23:28:28Z | its last upload, "2.25 MH/s, mining \| node 139757 blocks, 0 peers" at 23:28:28Z, is 22 s after `update-now-0311-d937c69d-37ba0461` was published; no job line reached the intake before the silence. Its 0.3.10 install kept it silent 55 minutes (21:41 to 22:36Z), so this is its install in progress until shown otherwise; publish 2 does not wait on it (a 2 MH/s machine whose 0.3.11 reads the nine fields when it returns; on 0.3.10 its node would die on the file until the forced apply, the C39 case for one machine) |
| PC 1 | unreachable tonight (section 3a); refused by every peer on 1f4b4425 until its morning relaunch takes 0.3.11 through the manifest | |
## 8. The rollout, step 2 (the object sweep to 0139ab9d...)
| Step | Time | Result |
|---|---|---|
| 2a the floor | 23:55:42Z | tip DAA 140,706; 154,800 - 140,706 = 14,094 >= 10,800, so N4 = N5 = 154,800 stand (the floor holds until DAA 144,000); no re-pin, no rebuild |
| 2b the observer | 23:56:59Z (pid 97249) | `/tmp/igneum-devnet/override-v3.json` switched to the nine-field object; `igneumd/2.1.0-89dfcb95`, `Program class v3 from the override file: active from epoch 43 (DAA score 154800 ...)`, `Proving v1 from the override file: ... 154800, 8 blocks a segment, unproven after 600 DAA, aggregator share 1000 bps`, digest 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 |
| 2b node 1 | 23:57:12Z (pid 97417) | the same lines and digest; it refused the seed (still 4d8f8bb6...) at 23:57:13Z and registered it at 23:57:42Z (protocol version 15), 29 s after the seed's own restart |
| 2b the seed | 23:57:30Z (MainPID 126124) | `/etc/igneum/override-v3.json` the nine-field object; the same binary 63cf490d..., the same lines, digest 0139ab9d... |
| 2b the manifest | 23:57:49Z | `publish-manifest.sh --version 0.3.11 --override '<the nine-field object>' --activation-height 154800 --deadline-note "program class v3 + proving v1" --notes '<section 1>' --public --deploy`; the live manifest's `consensus` read back byte-identical at the edge (the nine fields, `activation_height` 154800) |
| 2b update-now, the Mac and the laptop | 00:00:30Z (`update-now-0311-switch-d937c69d-37ba0461`, apps woken) | the Mac ran it 00:01:06Z: `0.3.11 is current`, `consensus parameters from the signed manifest: {... nine fields ...}`, `consensus override changed (.../Igneum/app/override.json); the node restarts with it at a safe moment`; the Mac's node card is node 1 (external: pid 0, starts 0, `consensus_digest` empty), already restarted by hand at 23:57:12Z, so there was nothing for the app to restart and its digest is node 1's. The laptop (PC 37ba0461) was not on the air (below) |
| 2b update-now, PC 2 | 00:01:10Z (`update-now-0311-switch-1ccfe586`, on the Counter ASIC coordinator's "PC 2 clear": `floor-restore-1` closed 23:57:29Z) | PC 2 fetched the woken file 00:01:41Z, ran the job the same second (`0.3.11 is current`, `consensus override changed (C:\Users\Admin\AppData\Local\igneum\app\override.json); the node restarts with it at a safe moment`), its node restarted at 00:01:42Z (`Consensus params digest: 0139ab9d...`, `igneumd/2.1.0`), the first STATUS 00:02:01Z "0.00 MH/s, waiting, node 141065 blocks, 1 peers, synced", mining at 00:02:31Z, 106.48 MH/s with 830 accepted this run and 0 faults at 00:06:31Z; the run id stays `win-1ccfe586-20261005-235130` (the node restarted, not the engine) |
| the window | 23:57:42 to 00:01:42Z | PC 2 (on 4d8f8bb6...) refused the seed (on 0139ab9d...) at 23:57:42Z and 23:58:12Z and mined on the old side alone; node 1 refused the seed once (23:57:13Z). The new side (the observer, node 1, the seed) had no miner on it either: the Mac's miner was off node 1 from 23:57:14Z (the next row), so the new side only relayed PC 2's old-side blocks it had already accepted (`PoW accepted ... daa 140757` at 23:57:32Z was the last) and waited; the two sides rejoined when PC 2's node restarted at 00:01:42Z and PC 2's hash carried the chain. The observer's tip read 140,757 at 00:02:22Z and 141,645 at 00:10:39Z (about 1.7 blocks/s, the catch-up after the rejoin), no stall as in step 1 |
| the Mac's miner (a miner finding) | 23:57:14Z to 00:08:34Z | the miner (`igneum-miner mine grpc://127.0.0.1:26610`) lost node 1 at node 1's 23:57:12Z restart and NEVER reconnected: 5,317 `submit error ... Not connected to server` and `template error: Not connected to server` lines, its TEMPLATES line frozen at `templates=2350` with `subscribed=true` while `fetch_errors` climbed (143 at 23:59:34Z, 595 at 00:05:07Z), the card's line "the node is not answering; the miner retries", the STATUS line still "26 MH/s, mining" with the accepted count frozen at 18,117 (773 this run). The retries are template and submit calls on a dead gRPC channel; nothing re-subscribes. Recovery through a job: `publish-jobs.sh add --kind restart --target d937c69d --what miners` (`restart-miners-0311-d937c69d`, 00:08:01Z); the Mac ran it 00:08:31Z (`miners restarted`), the new miner (pid 8567) started 00:08:33Z, its first block was accepted 00:08:34Z, the kernel race settled at 26.9 MH/s 00:09:09Z. Lost: 11 min 20 s of 26 MH/s. In step 1 this was masked: node 1's 23:24:06Z restart was followed by the Mac's own engine restart (the update) at 23:29:05Z, and the Mac was paused anyway |
| PC 2 at the epoch boundary | 23:52:22Z | 40 s after PC 2's first 0.3.11 pack the hour turned (`f4d9d3d8...` to `cb5b51cc...`): `hourly program changed: the pair was not prepared (unexpected seeds); the worker compiles inline`, 80 s of `worker fault: seed mismatch: the worker holds another program; preparing the current pair epoch cb5b51cc... day 2`, the worker on the new epoch at 23:53:42Z (113 MH/s), one `did not come back within 180 s` line for the old worker instance at 23:56:01Z. The restart landed inside the minute before the boundary, before the next epoch's prepare had run; the attempt-aware path (`program pack checked ... attempt 0`) recovered it without a job. Also at 23:52:14Z: `prover: block 74094 shard 0: ... Failed to create the CUDA prover impl: ... PermissionDenied (a GPU-server socket ...)`, the 0.3.10 socket class again on the first shard after the update; shards after it proved (`block 93665 shard 0 paid 0.9446373 IGN` at 00:06:09Z) |
| the laptop (PC 37ba0461) | absent | no upload since 23:28:28Z (section 7) and no card on the console at 00:08Z; both update-now jobs wait for it in the jobs file; its 0.3.10 app takes 0.3.11 and the nine-field object through the manifest when it is next on the air |
| PC 1 (ae432dc7), Sam's Mac (3a9bf309) | pending their relaunch | PC 1 down since 22:31:06Z (section 3a), Sam's Mac quit since 20:47Z on 0.3.9; each takes the manifest at its relaunch: the OLD app writes the nine-field object at the manifest take (`ota.rs` 661) and its old node, if restarted with that file before the 0.3.11 install lands, dies on the unknown fields (`deny_unknown_fields`, the reviewer's C39); the install then starts the 0.3.11 node on the file, so the worst case is one node death inside the update, to be read in the morning's intake |
## 9. The digest sweep
| Node | Binary | Digest | Since |
|---|---|---|---|
| the observer (`/tmp/igneum-devnet/observer-v4`, rpc 26640, json 28640) | `igneumd/2.1.0-89dfcb95` (bd7f043c...) | 0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888 | 23:56:59Z |
| node 1 (`/tmp/igneum-devnet/node1`, rpc 26610) | the same | 0139ab9d... | 23:57:12Z |
| the seed (188.245.5.161, `igneumd.service`) | `igneumd/2.1.0-89dfcb95` (63cf490d..., glibc 2.36 target) | 0139ab9d... | 23:57:30Z |
| PC 2 (1ccfe586, `devnet-v4`, rpc 26610) | the 0.3.11 installer's `igneumd.exe` be8e83c0... (PC 2's own build job) | 0139ab9d... | 00:01:42Z |
| the Mac (d937c69d) | attached to node 1 (no node of its own) | node 1's | 23:57:12Z |
| the laptop (37ba0461) | 0.3.10 node, 1f4b4425... at its last upload | pending | silent since 23:28:28Z |
| PC 1 (ae432dc7) | 0.3.10 node, 1f4b4425... | pending | down since 22:31:06Z |
| Sam's Mac (3a9bf309) | 0.3.9 node a24ab01a | pending | quit since 20:47Z |
The chain at 00:10:39Z: tip DAA 141,645 on the observer (age 1.4 s), node 1 with 2 peers (the seed and PC 2's side), PC 2 with 1 peer at
106 to 109 MH/s, the Mac at 26 MH/s, every live node on 0139ab9d... Epoch 43 (DAA 154,800) is about 3 h 45 min out at 0.965 blocks/s
(about 03:55Z); the class v3 and proving v1 lines of every node name it.
## 10. The next cut
| Branch | What | Why not 0.3.11 |
|---|---|---|
| `ember-tune` e600e63 | the C35 guard: `Engine.no_ota` (`IGNEUM_APP_NO_OTA=1` or `--sweep` skips the OTA tick and refuses Check now, logged at start), 6 lines in `engine.rs`, separable like the quit-source hunk; the playbooks and `tools/ci/second-engine-check.sh` (file not pipe, tree ended, NO_OTA set) ride with it. Without it any test engine on an old Cargo version can install over the fleet's app (section 3a) | arrived after the merge |
| `ember-tune` b671c8b (and the tune behind it) | the C35 fix: `Cmd::Quit(&'static str)` so every "quit:" line names its sender (the window host's stdin, the host gone, `POST /api/quit`, the sweep's end), `elevation_allowed()` = Power control alone (the unattended sweep on PC 1 raised one UAC prompt at 22:30Z under the old rule), no power cap at start under `--sweep`; the quit-source hunk is separable (main.rs 2 lines, server.rs 1 line, engine.rs the Quit arm, `elevation_allowed` and its test) | arrived after the tree closed at 23bc2b2 (the app, the DMG and PC 1's job carry it); not among the branches named for this cut |
| `fud-close` (the ledger closer's main branch, a22ba27a60a6f1c64; its ready tip was due about 23:05Z) | 45 public-text fixes on the site and litepaper, spec 8.3 and 8.8, two CI checks, relay fixes; touches `packfile.h` and `host.c`, so taking it means the two Windows workers, the Mac worker and the DMG rebuilt from the merged tip (G5) | offered by the Counter ASIC coordinator at 22:5xZ after the tree closed; not among the branches named for this cut; the fork-side `ledger-fixes` is not in 0.3.11 either |
| `explorer` d7e797c (and 3e01212) | `/api/stats` gains `proving`; `tools/ci/public-api-check.mjs` then FAILS when the live API lacks it, and ci.yml runs that check against the live site on every master push, which reads the OLD API until Vercel redeploys after the push (the reviewer's C36) | not in 23bc2b2 (only on the explorer branch); its merge needs the check to retry for a few minutes or to require `proving` only when `observer_updated_at` is newer than the commit |
| the app's job queue | an update-now job pre-empts a running job instead of queuing behind it, or the queue reports its wait in the STATUS line (tonight PC 2's update waited 11 minutes behind a hung sweep's cap, section 7; the Counter ASIC coordinator's ask) | app change |
| `proving-v1` c36dfea | docs only, after the code tip 22c2363 | its agent's choice: docs follow |
| the 0.3.10 list (`release-0.3.10.md` section 11): fork `pack-loop` 05ef0fa3, `job-console`, the rest of `opencl-rdna4`, `opencl-rdna4-telemetry` | | unchanged |
## 11. Open after the cut
| Item | What | Owner |
|---|---|---|
| the miner's dead channel (section 8) | `igneum-miner mine grpc://...` keeps `subscribed=true` after the node restarts and retries template and submit calls on the closed channel for ever (11 min 20 s tonight, 5,317 errors, hash burned at 26 MH/s with 0 accepted); it must re-subscribe on the first `Not connected to server` and the app's card must read "the miner lost the node" instead of "mining, 26 MH/s". Until the fix: every hand restart of node 1 is followed by `publish-jobs.sh add --kind restart --target d937c69d --what miners` (the runbook's step), and the Mac's card is read by its accepted count, not its MH/s | miner (next cut) |
| the HiveOS package carries no override | `packaging/hive/h-run.sh` starts the rig's node without `--override-params-file`; a rig on `igneum-hive-0.3.11.tar.gz` runs on genesis params and is refused by every peer; the file must come from the Flight Sheet's extra config as `PEERS=` does (section 5) | packaging (next cut) |
| the prepared pack and a restart near the hour (section 8) | an engine restarted in the last minute before an epoch boundary has no prepared pair for the next hour (80 s of seed-mismatch faults on PC 2 at 23:52:22Z, recovered by the attempt-aware path); the prepare should run at engine start for the next epoch too, or the update-now path should wait out the boundary when it is under 2 min away | app |
| PC 2's first shard after an update | `Failed to create the CUDA prover impl: ... PermissionDenied (a GPU-server socket ...)` on block 74094 at 23:52:14Z, the 0.3.10 socket class (fixed 22:01Z for the running engine; the root socket outlives the engine restart); the later shards proved. The socket unlink belongs in the engine's prover start, not only in the playbooks (`prover-socket-check.sh` covers the playbooks) | app / proving |
| the relay machine named PC1 (section 3a) | it is the 1ccfe586 box: rename it (`node tools/relay.mjs name DESKTOP-KMCV30N PC2`), give the ae432dc7 box its own agent, and no relay task to "PC1" without the Counter ASIC coordinator's word until then | relay (the project lead in the morning for the PC 1 box) |
| PC 1 down since 22:31:06Z | relaunch in the morning (the installed app, double-clicked or through the Start menu, never elevated); it takes 0.3.11 and the nine-field object through the manifest, one node death inside the update possible (section 8); the intake's first lines of `win-ae432dc7-...` are the check; `ember-tune` b671c8b carries the C35 fix for the quit's sender | the project lead, then the release engineer |
| the laptop (37ba0461) | silent since 23:28:28Z, off the console; its 0.3.10 install kept it silent 55 min once already; when it uploads again its first STATUS line must show 0.3.11, the nine-field object in the manifest log line and 0139ab9d... | watch |
| Sam's Mac (3a9bf309) | 0.3.9, quit since 20:47Z; `min_supported` is 0.3.0 so it updates straight to 0.3.11 at its relaunch | watch |
| the app's job queue | an update-now queues behind a running job (PC 2's step-1 update waited 11 min behind a hung sweep's 30 min cap, section 7); pre-empt, or report the wait in the STATUS line | app (section 10) |
| the public index at the edge | `dl/public/igneum-downloads.json` alternates between the unsigned and the signed copy at the edge for minutes after a deploy, so the ship's verify refuses it once per cut (0.3.10 and 0.3.11 both); `--from console` after it settles is the workaround; the verify should accept either copy while both carry the cut's hashes, or the deploy should purge the edge | ship-app |
| the pre-push site flip | the pre-push hook rebuilds `site/` and leaves the tree modified (`git checkout -- site/` after every push); the site build is not idempotent | site |
| the explorer branch's strict check (C36) | `public-api-check.mjs` on `explorer` d7e797c fails against the live API until Vercel redeploys; merge it right after a master push, not before | explorer (section 10) |
| C1 (the reviewer) | the 0.3.10 `exportSegments` lacks `daaScore` and `feesV1ActivationDaa`; proving-v1's `rpc.rs:982` (eb32c645) carries them; the decision on which the aggregator reads is due 16:00Z 6 October | the Counter ASIC coordinator |
| the Metal worker at resume (G4b) | "the pair was not prepared" at the Mac's 23:39:10Z resume (section 7), compiled inline; same class as the PC 2 boundary row above | app |
| the 0.3.10 list | `release-0.3.10.md` section 11, unchanged: fork `pack-loop` 05ef0fa3, `job-console`, the rest of `opencl-rdna4`, `opencl-rdna4-telemetry`, the per-job zip pin check | |
## 12. The merge and the push
The plan's last branch commit 7ab72f3 (sections 8, 9, 11); `git merge --no-ff release-0.3.11` onto the local master 063baf6 (the CLAUDE.md
standing rule of 22:52Z, unpushed until now) = master 30cd292 (parents 063baf6, 7ab72f3), no conflict (the site files merged clean this time);
pushed 00:13:04Z (`b38f3de..30cd292`); the pre-push hook's site flip (bench.html, downloads.json, index.html, journey.json, miner.html)
restored with `git checkout -- site/`. CI on 30cd292: `ci` run 37392831312, `windows-ci` run 37392831184 (both started 00:13:13Z). No
release tag: 0.3.9 and 0.3.10 carry none (the repo's tags are backups and the CA2 history marks), so none is cut here without the word.
The fleet at the push: the Mac 25.7 MH/s with 55 accepted since its miner restart, PC 2 105 to 116 MH/s, every live node on 0139ab9d...,
tip DAA 141,715 at 00:12:36Z.

View file

@ -0,0 +1,193 @@
# Igneum Miner 0.3.12: the fresh-record rule switch (proving v1) and the app cut, prepared to the publish gate, 6 October 2026
Release engineer, from 08:05 UTC, on the coordinator's instruction ("prepare 0.3.12, APP ONLY, up to the publish gate and STOP there; the project lead
gives the go"), widened at 08:55Z on its clock: "no longer app only", the proving agent proved the segment record rule needs a consensus
switch, so the node fork `proving-v1` 0f0dda95 (`proving_v1_fresh_rule_daa`: never by default, in the digest only once set; from it a fresh
segment record is valid whenever the previous segment is not proven) and the app's segment-aligned prover (272b025, docs aea2f6a) ride in it.
Worktree `/Users/joshm/Projects/igneum-wt-ship0312`, branch `release-0.3.12` from master ddfcdac; `vendor/` symlinked to the main checkout's (46
entries); the fork worktree `vendor/igneum-node-0312`, branch `release-0.3.12-node` = 0f0dda95 cherry-picked onto 89dfcb95 (its parent ece42979
is inside 89dfcb95, so the rebase is the one commit: `params.rs`, `exec/proving.rs`, `exec/rpc.rs`, `daemon.rs`) = **83089544**. The 0.3.11 recipe (`release-0.3.11.md`) throughout; every Mac build under the main checkout's lock;
igneum-labs commits. Times are UTC.
## 1. What 0.3.12 carries
| Change | Where | State |
|---|---|---|
| `/api/state` never answers `{}` again: `paid_wei` (u128) is a decimal string, the error reply is logged; test | `proving-v1` app 6714a45 (the first item) | merged 9bcf4cd (the `docs/bench-log.md` conflict: both sides kept, the log is append-only) |
| An update published over an hour before the engine started skips the hourly rollout slot; `manifest::unix_from_rfc3339` + tests | `update-catchup` 2207cd7 | merged b984c17 |
| The GPU list ordered by performance (usable, discrete before integrated, rate in 5 MH/s buckets, memory); 3 UI tests | `card-order` ffb2bfa | merged c6608c1 |
| HiveOS local mode carries the override (`OVERRIDE=` in the Flight Sheet's extra config, written to `data/override-params.json` by `h-run.sh`, C41), rigs mine only until a Linux prover ships, `IDENTITIES=auto` by VRAM, the per-card README table | `hive-words` 98271ff (packaging/hive only) | merged b0a6231 |
| Ember Tune: two-knob plans + priors + the UI line; every quit names its source (b671c8b); a second engine never runs the updater (e600e63, C35); no pipe into a second engine (8ab9068); `jobrun.rs` elevated `follow_file` (1e9550e); the BOM fix + CI check (8273494); Power control switch (49bbe14 = 3562f26); `igneum-gpu-telemetry.exe` (ADLX) built by `build-windows.sh` and carried in the Windows inputs | `ember-tune` 9a6469f | NOT MERGED: conflicts in seven files against the 0.3.10/0.3.11 app (its base ca8d9f3 predates both): `ci.yml`, `config.rs`, `engine.rs` (the detect path, the power-cap plan, the test module), `ui/app.js` (four hunks against miner-ui-2's View), `ui/index.html` (the settings panel 0.3.10 removed), `proto-opencl/README.md`, `bench-log.md`. Its agent is rebasing it onto release-0.3.12 (section 2) |
| The hidden-console builder for every elevated launch, `windows-spawn-check.mjs`, the PC 1 console-watch scripts | `job-console` 13755b9 (+ 3562f26 Power control) | NOT MERGED: conflicts in six files (`ci.yml`, `config.rs`, `engine.rs`, `jobrun.rs`, `app.js`, `index.html`), base a93199a; carried by the ember-tune rebase (it already holds 3562f26) |
| The miner's gRPC resubscribe after a node restart (C43, the 0.3.11 finding) | no commit exists (the ledger entries b19fe5f, 0751dde, c8c831c only) | OWED, listed in section 10 |
| The fresh-record rule switch: `proving_v1_fresh_rule_daa` (Option, never by default; a node with the field set prints it and carries it in the digest; a fresh segment record is valid from it whenever the previous segment is not proven) | fork `proving-v1` 0f0dda95 on ece42979 | cherry-picked onto 89dfcb95 as 83089544 (`release-0.3.12-node`) |
| The segment-aligned prover: a segment record the chain rule refuses is held and offered again every pass until the segment closes; the fast-time harness on the fresh-record rule (both cases); the prover host and export; the WSL2 prover package script; `infra/fast-time/override-60x.json` (measured by its agent: 9 segments per 30 min on PC 2, 72 of 72 shards paid, 11% hash cost, 17.6 GB peak) | `proving-v1` app 272b025 + docs aea2f6a (on 6714a45) | merged 49e0e2c (clean) |
| The packaged line (C34): the ten-field object of section 4 | 7dd3ff7 | `packaged-config.sh --test` passes |
| The six version files | 81e4ecb (`--check`: 0.3.12 in all 6) | |
Left out on the coordinator's word: prover-floor's server (its packaging row is 0.3.13), explorer d7e797c, pool-v0, rig-install, ota-k2, the
ledger forks.
Changelog line (draft, for the manifest notes at the go): "Igneum Miner 0.3.12: the fresh-record rule for proving v1 from DAA 192,000 (a fresh
segment record is valid whenever the previous segment is not proven) and the segment-aligned prover; the GPU list in performance order; an
old update no longer waits for the hour; Ember Tune (every card tuned for MH per watt, Power control off by default, the app never asks for
administrator rights on its own); a second engine never installs over the app; /api/state always answers; HiveOS rigs carry the override.
Node 83089544."
## 2. The branch
| Commit | What |
|---|---|
| 9bcf4cd, b984c17, c6608c1, b0a6231 | the four merges above, in the coordinator's order (proving-v1 first) |
| 81e4ecb | `Igneum Miner 0.3.12: the six version files` |
| ebea8b6 | the ember-tune rebase tip 7f6c4e6 (with job-console 13755b9 inside), merged as one branch (section 3) |
| 11e8ca6, ab01f48, 01abcc2 | the plan |
| 062c3f8 | `node-source.pin` 83089544 with the second inputs push (the Windows-build commit of 0.3.12) |
| 37b6a7f | `tools/proving-v1/pc2-agg-cost.ps1`: `pkill -f sp1-gpu-server` (the CI root-socket check) |
| 88df58e | master d3b64cb merged (docs only): the release tip, CI green |
| 6532adf | `make-payload.sh`: on CI the AMD telemetry helper is taken from the unpacked inputs (the worker glob `igneum-worker-*.exe` missed `igneum-gpu-telemetry.exe`, so the first payload, run 37435975425, shipped without it: the inputs had it, the zip did not). The CI commit of 0.3.12 |
Checks on 81e4ecb before the rebase landed: the app `cargo test --release -p igneum-app` under the lock: ok 115 (lib) + 28 (ota-sign) + 8
(prove-verify), 0 failed (08:19:22 to 08:19:28Z, warm target cloned from the 0.3.11 worktree); the UI tests `notices`, `update-card`, `view`:
23 of 23.
## 3. Builds and artefacts
| What | Command | Result |
|---|---|---|
| The HiveOS package (first build, app-only cut) | the 0.3.11 node and workers with hive-words' scripts | 08:19:45Z: b0a20917... (24,501,272); superseded below once the node changed |
| The merged tree | ember-tune 7f6c4e6 (release-0.3.12 b0a6231 merged INTO ember-tune as c5918c7, job-console 13755b9 cherry-picked on top; 0.3.11's six-section View and card order kept whole, Ember Tune's TuneLine block and the Power control switch added in 0.3.11's markup, `engine.rs` keeps the detect arm with the tune fields in `hotplug::apply_pref`, both test modules, `jobrun.rs` the hidden-console builder plus `follow_file`) merged as one branch | **ebea8b6**, 08:22Z (the CI commit is 062c3f8); the version files still 0.3.12 in all 6; packaged line, `node-source.pin` and `vendor/` untouched against master |
| The app | `cargo test --release -p igneum-app` under the lock, then `cargo build --release` | 08:22:56 to 08:23:06Z: ok 133 + 28 + 8, 0 failed; `igneum-app 0.3.12` (2,273,664) |
| The UI and relay tests | `node --test` notices, update-card, view, tune-line; `relay/test/*.test.mjs` | 26 of 26; 23 of 23 |
| The CI checks on the Mac | identity, no-conflict-markers, copied-sources, signer-pipe, prover-socket, second-engine, bash-body (self-test + tree), kit-path (self-test + tree), windows-spawn (self-test + tree), pinned-guests, no-secrets, check-workflow-shell | all ok; `link-check` passes after `node site/build.mjs` (as ci.yml runs it: the committed `litepaper.html` points at `/bench#counter-asic-2-0-the-numbers`, an id the site build creates from `bench-log.md`) |
| The Windows workers and the AMD telemetry helper | `proto-cuda/nvrtc/build-windows.sh` (mingw) under the lock, 08:23Z | the worker SOURCES are unchanged against master (`git diff master HEAD -- proto-cuda proto-opencl proto-metal igneum-pow`: only `build-windows.sh`, the new `gpu-telemetry.c` and its `.rc`), so the inputs carry the 0.3.11-verified workers that mined all night, igneum-worker-cuda.exe 2b3b8c92885442179f6bf2907c6f3eb453dc4a19908d90fd05981a09b7c2674c (1,536,512) and igneum-worker-opencl.exe edc4a75da3b93d814caa69fd635010780d63d5b622ec24c3741d433c584f91e3 (478,208), not this morning's rebuild of the same sources (12bfaa27..., e0fd7042...: mingw PE builds are not byte-reproducible); NEW igneum-gpu-telemetry.exe 8d679b52b19af3cbd6bf4fd6f77d337b2fb78af13d02627e7aa7e92507993459 (387,584; ADLX, SetupAPI, PDH; the Igneum resources, version 0.3.0 as the workers carry) |
| The Windows inputs | `IGNEUM_WIN_RELEASE=<fork>/target-integration/x86_64-pc-windows-gnu/release IGNEUM_NODE_SRC=vendor/igneum-node-0312 packaging/windows/push-inputs.sh`, 08:24:29Z (a deploy of the downloads folder only; the manifest untouched) | node 89dfcb95 (igneumd.exe be8e83c0..., igneum-miner.exe 1ba1a249..., PC 2's 0.3.11 build), the two workers above, the telemetry helper, the mingw DLLs and nvrtc; signed, verified, live (HTTP 200); `node-source.pin` unchanged 89dfcb95 |
| The DMG (first build, app-only cut) | the 0.3.11 node | 08:24:33Z: ff630e9d... (41,630,620); superseded below once the node changed |
| The fork's Mac node, 83089544 | `CARGO_TARGET_DIR=vendor/igneum-node/target-0312 cargo build --release -j 4 -p kaspad -p igneum-miner --features kaspad/igneum-pow` from `vendor/igneum-node-0312`, under the lock; the target dir cloned by APFS from `target-0311`; `target-integration` in the fork worktree links to it | 08:38:13 to 08:41:31Z: igneumd **746a931fde9b840ca444a03cd757854e2e2ce7ebc4782ddd00ed161644d705f9** (41,386,160), igneum-miner 5381683e5717d91416c5a97456e0d050dd9645b1e27bce7d55808ce621cc1a26 (8,763,936); `igneumd/2.1.0-83089544` |
| The seed's Linux node (glibc 2.36 target, zig) | `NODE_SRC=<abs fork> TARGET_DIR=vendor/igneum-node/target-0312-linux OUT_DIR=<scratch>/cross infra/cross/build-linux.sh` under the lock | 08:38:21 to 08:41:18Z (175 s): igneumd **4f142d5148f218f1286e24e7c4a167aa2f54262336f96e7cf281f520c714fc6f** (47,919,144), igneum-miner 38397ae66c265b63db8e5458b46e7feb942121a7dc5625919df0a8d35e7a1ba1 (9,861,072); version.txt names 83089544 |
| The prover host and export (the pinned guests unchanged) | `cargo build --release -j 4 -p igneum-prove-export -p igneum-prove-host` in `proving/igneum-prove` under the lock (the target cloned from the 0.3.11 worktree) | 08:39:07Z: igneum-prove-host b90d58d0529ce29f0e7ca8ae780a6442f92edcf1c71752fc60fbc72bc5c11fd8 (58,626,560), igneum-prove-export b60056127d32bda363c0e305e73e9f699a5774c0988aee5bfd28a0fc61a56b9a (2,808,160); `--mode id`: shard program id 0x2b1a81cb413236cf063077b46ed3111628f6c41036bcf6e23ee4cbbf5679ef7a, the pin of 0.3.9 to 0.3.11; `pinned-guests-check` ok, `proving/igneum-prove/elf/` untouched |
| The HiveOS package | `NODE_OUT=<scratch>/cross WORKERS_OUT=<the 0.3.11 Linux workers> VERSION=0.3.12 OUT=<scratch>/hive packaging/hive/make-hive-package.sh` (the Linux workers 4aaff27f.../82d90890... unchanged: their sources are) | 08:41:54Z: `igneum-hive-0.3.12.tar.gz` **7972af92e7cd9a032303eca4d95b533f53e0e68d1b9cae5bfe406a5b7c30a454** (24,506,282); `h-run.sh` writes `data/override-params.json` from `OVERRIDE` and starts the node with `--override-params-file` (the 0.3.11 open item closed); the node inside is 83089544 |
| The DMG | `NODE=<fork igneumd> MINER=<fork igneum-miner> PROVE_HOST/PROVE_EXPORT=<this tree's build> packaging/mac/build-dmg.sh` under the lock | 08:42:49Z: `Igneum-Miner-0.3.12.dmg` **7a4a5f5f772956e983127280a5ec62a4fcfaf903b3afa38fe3a89a37cee23520** (41,702,535), engine 0.3.12, node 83089544 (igneumd 41,163,744 inside, stripped by the DMG build), the new prover host and export, `igneum-bench` from `proto-metal/main.swift` (unchanged, 66ec0e78...), `packaged-config` carries the ten-field object, hdiutil checksum valid |
| The node suites with the igneum-pow feature (the coordinator's ask; the PC runner's test units carry no features field, so this is the Mac's run; the PC 2 run is owed to the Counter ASIC 3.0 coordinator's window, section 3a) | `CARGO_TARGET_DIR=vendor/igneum-node/target-0312 cargo test --release -j 4 -p kaspa-consensus -p kaspa-consensus-core -p igneum-exec -p kaspa-pow -p igneum-miner -p kaspa-p2p-flows --features kaspa-consensus/igneum-pow,kaspa-pow/igneum-pow` from the fork, under the lock, 08:39:31 to 08:45:01Z | igneum-exec 17 of 17, igneum-miner 18 of 18, kaspa-consensus 97 passed, 2 failed, 4 ignored. The two: (1) `pruning_proof::igneum_m20_tests::witnesses_are_checked_in_epoch_order_under_their_own_seeds` (`igneum_m20_tests.rs:122`: the expected `EpochSeeds.era` is all zeros, the code draws `515e...`: the test predates the era draw of class v3) FAILS THE SAME on 89dfcb95 (run 08:45:48Z on the 0.3.11 fork): the known M20 era fail, NOT fixed by 0f0dda95, still owed; (2) `finality::tests::ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list` (`finality.rs:1883`) failed inside the full crate run and PASSES alone on both 83089544 and 89dfcb95: order-dependent, not a regression of this cut, owed as flaky. kaspa-consensus-core 107 passed, 1 failed (`config::params::tests::fast_time_60x_file_is_the_devnet_at_60x`: `infra/fast-time/override-60x.json` does not parse into `OverrideParams`, "duplicate field `proving_v1_activation_daa`" at line 64: the file has carried a second proving-v1 block since c2544be on 5 October, so the test fails on master's file and on 89dfcb95 alike; not a 0.3.12 regression, the file is owed a dedupe), `db_compat` 7 of 7, kaspa-pow 14 of 14, kaspa-p2p-flows 33 of 33 (08:47:13Z, no fail-fast). Net: 3 failures, each present on 89dfcb95, none from 0f0dda95 |
| PC 2 build-and-suite job | `IGNEUM_WIN_RELEASE=<scratch>/pc2-out node tools/build-job.mjs run --node vendor/igneum-node-0312 --target 1ccfe586 --targets linux,windows --node-tests "kaspa-consensus kaspa-consensus-core igneum-exec kaspa-pow igneum-miner kaspa-p2p-flows" --app-tests igneum-app` from this worktree, published 08:54:56Z on the prover-floor agent's "PC 2 is yours" (its floor-core-alone and floor-core-miner closed 08:48:37Z and 08:53:13Z, after the Counter ASIC 3.0 coordinator's release at 08:43:24Z); CPU only, the prover on, nothing else touched. The coordinator's later order (after the prover-floor agent's SECOND pair) arrived once the job had run; the app's queue serialised them anyway: this job ended 09:02:04Z and floor-build-6 started 09:02:05Z, then floor-core2-miner closed 09:13:22Z with the prover on and the miners never stopped, so nothing ran beside a GPU row | `build-20261006-085456`: started 08:55:24Z, done 09:02:04Z (400 s), every stage ok: Linux node 135 s (igneumd 34877b86..., igneum-miner c779777f...), Windows node 165 s (igneumd.exe 5bbcbd59f592fa31bf31c18516cef81cc0e7e537d398382fc8063e3402d80917, igneum-miner.exe af973318...; PC-built, NOT shipped: the inputs carry the Mac cross-build f580b4aa..., placed under the scratchpad), the app both targets; `RESULT test node [the six crates] exit 0 44 s` (without the igneum-pow feature, the runner's shape: the M20 era test and the fast-time file test are outside its reach there) and `RESULT test app/igneum-app exit 0 6 s` |
| The Windows node exes | `CARGO_TARGET_DIR=vendor/igneum-node/target-0312-win proto-cuda/windows-node/cross-build.sh <fork> 4` on the Mac (mingw, the 0.3.5/0.3.6/0.3.9 path; PC 2 is the Counter ASIC 3.0 coordinator's this morning), the target cloned from `target-release-win`, under the lock | 08:40:43 to 08:48:05Z (6 min 42 s): igneumd.exe **f580b4aad1e19a47742d0d836a56dad36b9380d3890ca115b1babced4d83a8db** (52,177,920), igneum-miner.exe 06c17d4c23c8b1327793bebcd9b2cba115045ea91b68baea8cb92b232a94678b (11,040,768); static (KERNEL32, advapi32, api-ms-win-core only) |
| The Windows inputs, second push | `IGNEUM_WIN_RELEASE=<target-0312-win>/x86_64-pc-windows-gnu/release IGNEUM_NODE_SRC=vendor/igneum-node-0312 packaging/windows/push-inputs.sh`, 08:48:28Z | node 83089544 (the two exes above), the 0.3.11-verified workers 2b3b8c92.../edc4a75d..., the telemetry helper 8d679b52..., the mingw DLLs and nvrtc; `payload-inputs.zip` f8e567bd164b382d32a33fb488df658fa68555092ac6bdac85387d0a8bd5d547 (65,259,161), signed and verified, live (HTTP 200); `node-source.pin` 83089544 committed as **062c3f8**, the CI commit of 0.3.12 |
| The Windows installer and zip, first runs | runs 37435975425 (ebea8b6, no telemetry helper) and 37436904041 (6532adf, the 0.3.11 node) | superseded |
| The Windows installer and zip | `windows.yml` run 37438673235 on 062c3f8 (dispatched 08:49:12Z after the second inputs push) | `Igneum-Miner-Setup-0.3.12.exe` **f11a296acf1ea3efa8a6151efa357cc5c222e3b2ffec5701ea4bcefe29307810** (45,270,093); `igneum-windows-app.zip` **a6f33ef21bb1d1682f48ca22332d50c0302f4b4f552a99157581f3249c42ea3b** (65,507,597): igneumd.exe f580b4aa... (the cross-build, 52,177,920), igneum-miner.exe, igneum-app.exe 0.3.12 (3,700,736), the two workers, `igneum-gpu-telemetry.exe` (387,584) this time, the mingw DLLs, nvrtc64_120_0.dll and nvrtc-builtins64_128.dll |
## 4. The override objects and the digests (the 0.3.12 Mac node 746a931f..., ports 60975/60976, 22 s each, under `run`)
| Override file | Lines | Digest |
|---|---|---|
| none | `igneumd/2.1.0-83089544`, no activation line | c562d70e1428c9789823cc40067623b4767f7c555ce7ff4ea11c1498f013ef6c, EQUAL to 0.3.11's no-file digest: the new field is never by default and leaves the digest alone until set |
| the fleet's live nine-field object | the six activation lines of 0.3.11, no fresh-rule line | **0139ab9dc2992d449ec787d8f021974933631eb55740ab4b6ce9d5c226e72888**, EQUAL to the fleet's digest today: publish 1 (the binary) changes no handshake, a 0.3.12 node and a 0.3.11 node on the nine-field file accept each other |
| the ten-field object at the FIRST pin (`proving_v1_fresh_rule_daa` 192000, void: the floor failed at the go) | the six lines plus the fresh-rule line at 192000 | bd786a4b521e87c05bce3da4c46b4f4696deb16dfbdc913f181c980a8eb51688 (never published) |
| the ten-field object as SHIPPED (`proving_v1_fresh_rule_daa` 198000) | the six lines plus `Proving v1 fresh-record rule from the override file: from DAA score 198000 a fresh segment record is valid whenever the previous segment is not proven` | **7bd98cc4118616455709d5e32a30b799e6e67caa42d2b5d09875cd49848a7ed7** (read 11:22:01Z on 746a931f...) |
The ten-field object (the packaged line 7dd3ff7, the manifest of publish 2, the hand nodes' and the seed's files at step 2):
```
{"difficulty_v2_activation_daa":33000,"proving_v0_activation_daa":84100,"fees_v1_activation_daa":210000,"finality_v3_activation_daa":135200,"program_class_v3_activation_daa":154800,"proving_v1_activation_daa":154800,"proving_v1_segment_blocks":8,"proving_v1_unproven_daa":600,"proving_v1_aggregator_share_bps":1000,"proving_v1_fresh_rule_daa":198000}
```
N was first pinned at 192,000 (tip + 14,400 for a publish near 10:15Z; the floor would hold while the tip was at or under 181,200, about
11:15Z). the project lead's go came at 11:20Z with the tip at 181,582: the floor read 10,418, under 10,800, so N was RE-PINNED to 198,000 (tip + 14,400
for publish 2 near 11:55Z; the floor holds until tip 187,200, about 12:55Z): the packaged line 8a6b133, the DMG rebuilt 11:22:08Z
(7bcbb8a94038ea7a87ebfab514b6771f93b8fce2d991340bd5610713dc9548e5, 41,702,608), the installer rebuilt on CI (windows-ci 37455874734 on 8a6b133,
green 11:27:33Z: `Igneum-Miner-Setup-0.3.12.exe` a6b3ea275373411e9f988ecd4ecc79f2cda5f68d54d681662dcbf7590689aef0, 45,275,988; `igneum-windows-app.zip`
7ab28e772670dc58428cda4c9ff584b35f70507057d525a85d04391ee145dc53, 65,507,595), the digest re-read, publish 1 delayed by 8 minutes. The lesson for
the next cut: pin N at the go, not at the forecast, or pin with a 3,600 margin over tip + 14,400 when the go is more than an hour out. A 0.3.11 node given the ten-field file dies on the unknown field
(`deny_unknown_fields`), which is why publish 2 comes only after every node runs the 0.3.12 binary (the reviewer's C39, the 0.3.11 order).
## 5. The publish gate: what runs at the project lead's go, in which order (the 0.3.11 two-publish shape)
This was the plan at the gate; sections 6 and 7 record what ran. Runbook: the session scratchpad's `r0312/rollout-0312.sh` (every step a function; `step_floor` before each publish).
| Step | What | Gate |
|---|---|---|
| 0 the floor | `step_floor`: 192,000 minus the tip's DAA at least 10,800 | read before publish 1 and again before publish 2 |
| 1a the hand nodes, the seed | the observer and node 1 on the 0.3.12 binary with the NINE-field file (`IGNEUMD=<fork>/target-integration/release/igneumd IGNEUMD_COMMIT=83089544 infra/devnet/restart-hand-nodes.sh '<nine>'`), then the seed (`IGNEUMD_LINUX=<scratch>/cross/igneumd IGNEUMD_LINUX_SHA256=4f142d51... infra/devnet/restart-seed.sh '<nine>'`); every one prints 0139ab9d..., nobody is refused; then `step_mac_miners` (the Mac's miner does not reconnect to a restarted node 1 by itself, C43) | the project lead's go |
| 1b publish 1 | `node tools/ship-app.mjs 0.3.12 --node vendor/igneum-node-0312 --branch release-0.3.12 --public --activation-height 154800 --deadline-note "program class v3 + proving v1" --notes '<section 1>' --from ci`: ci "already" (the green Windows run), fetch, dmg "already", copy, manifest with `consensus` CARRIED OVER (the nine-field object; the digest stays 0139ab9d...), deploy, verify (`--from console` after the public index settles at the edge), the console item; `--public` carries the HiveOS package 7972af92... | after 1a |
| 1c update-now | the Mac (d937c69d) first; the laptop (37ba0461) with it if it is on the air; PC 2 (1ccfe586) on the Counter ASIC 3.0 coordinator's word (PC 2 is its this morning; the proving agent's constraints: the app's prover stays on, no quit or restart of anything but the update's own); PC 1 (ae432dc7) last, once the project lead has relaunched its app (down since 22:31:06Z yesterday, on 0.3.10: it takes the nine-field object and 0.3.12 at its relaunch through the manifest; Power control is off by default so nothing asks for administrator rights) | each machine's STATUS line back on 0.3.12 with 0139ab9d... |
| 2a the floor again | `step_floor` | >= 10,800 or re-pin |
| 2b the hand nodes, the seed | the same two scripts with the TEN-field file; each prints bd786a4b... and refuses the nine-field side until it switches; `step_mac_miners` again | every app node on the 0.3.12 binary (1c) |
| 2b publish 2 | `publish-manifest.sh --version 0.3.12 --override '<ten>' --activation-height 192000 --deadline-note "proving v1 fresh-record rule" --notes '<section 1>' --public --deploy` | after the hand nodes |
| 2b update-now (switch) | the Mac and the laptop, then PC 2 on the 3.0 coordinator's word, then PC 1: each app writes the ten-field file at the manifest take and restarts its node at a safe moment (the Mac's node is node 1, already switched: nothing to restart) | |
| the sweep | every node prints bd786a4b521e87c05bce3da4c46b4f4696deb16dfbdc913f181c980a8eb51688; the fresh-record rule arms at DAA 192,000 | |
One line for the project lead, per machine, when he says go: the Mac and PC 2 each restart their engine once for 0.3.12 (under a minute, the miner back on
the next template) and their node once more for the fresh-record switch (a few seconds, mining resumes on the same chain); PC 1 does the
same at its relaunch and, with Power control off by default, never asks for administrator rights again (its 5090 runs uncapped until he
switches Power control on in Settings); the observer, node 1 and the seed are restarted by hand twice; from DAA 198,000 (about 15:55Z) a
prover may file a fresh segment record whenever the previous segment is not proven, so paid segments stop stalling behind an unproven one;
until the switch nothing changes in consensus (digest 0139ab9d... through publish 1).
## 6. The rollout (the project lead's go 11:20Z through the coordinator; two publishes)
Baseline 11:20:22Z: tip 181,582; the observer, node 1 and the seed on 89dfcb95 at 0139ab9d; the Mac app 0.3.11 (its miner PAUSED since
07:10Z on the project lead's order "stop mining on the Mac", not the C42 class: I resumed it once at 11:26:11Z before the order reached me and the
coordinator re-paused it; it stays paused, no restart-miners after the hand restarts); PC 2 0.3.11 at 113 MH/s; PC 1 0.3.11, relaunched by
the project lead at 11:17Z with the 5090 and a 4070 in the enclosure; the laptop and Sam's Mac off the air.
| Step | Time | Result |
|---|---|---|
| 0 the floor | 11:20:22Z | 10,418 < 10,800: FAILED at 192,000; re-pinned to 198,000 (section 4), publish 1 delayed to the installer rebuild |
| 1a the observer, node 1 | 11:22:12Z (pid 92464), 11:22:24Z (pid 92624) | `igneumd/2.1.0-83089544` on the nine-field file, digest 0139ab9d... (unchanged, nobody refused) |
| 1a the seed | 11:22:45Z (MainPID 136418) | the same binary 4f142d51..., the same digest |
| 1a restart-miners, the Mac | 11:22:56Z | ran; nothing to restart, the miner is paused on the project lead's order (above) |
| 1b publish 1 | the ship 11:28:30 to 11:31:55Z from cf1ad2b (master f11b02e merged first: the preflight refuses a tree behind origin/master) | ci "already" (37455874734), fetch "already" (the re-pinned installer), dmg "already", copy ok, manifest 0.3.12 with `consensus` CARRIED OVER (the nine-field object, activation 154800), deploy ok, verify refused the public index at the edge (every cut); `--from console` 11:44:41Z: item #368 |
| 1c update-now, the Mac | 11:32:18Z | engine restart 11:32:56Z (run `mac-d937c69d-20261006-113256`), "updated to Igneum Miner 0.3.12 from 0.3.11", STATUS "0.00 MH/s, paused, node 5 peers, synced" (node 1) |
| 1c update-now, PC 2 | 11:32:46Z (the 3.0 coordinator's mkdir lock `/tmp/igneum-devnet/pc2-ca3.lock` absent; `pc2-ca3.clear` is a note, not a lock) | engine restart 11:33:41Z (run `win-1ccfe586-20261006-113341`), igneumd 83089544 started 11:33:45Z on the nine-field file (0139ab9d), worker ready 11:34:39Z, mining 11:34:42Z, 0 faults |
| 1c update-now, PC 1 | 11:33:28Z (on the prover-floor agent's "PC 1 build closed" 11:31:34Z and the coordinator's "PC 1 back") | the installer downloaded and verified 11:34:06Z, "per-user install, no administrator prompt", engine restart 11:34:12Z (run `win-ae432dc7-20261006-113412`), cards "RTX 5090, RTX 4070 [discrete], AMD integrated [off]", STATUS mining 11:35:13Z, the 5090's race base 140.2 MH/s, digest 0139ab9d |
| 2a the floor | 11:36:53Z | tip 182,570; 15,430 >= 10,800 at 198,000 |
| 2b the observer, node 1 | 11:36:55Z (pid 13642), 11:37:07Z (pid 13777) | the ten-field file, digest **7bd98cc4118616455709d5e32a30b799e6e67caa42d2b5d09875cd49848a7ed7** |
| 2b the seed | 11:37:25Z (MainPID 136590) | 7bd98cc4... |
| 2b publish 2 | 11:37:36Z | `publish-manifest.sh --version 0.3.12 --override '<ten>' --activation-height 198000 --deadline-note "proving v1 fresh-record rule" --public --deploy`; the HiveOS package 7972af92... served at `/public/igneum-miner-hive.tar.gz` and `dl/public/igneum-hive-0.3.12.tar.gz` (HTTP 200, 24,506,282), the 0.3.11 package removed |
| 2b switch, the Mac | 11:40:30Z | ran 11:40:58Z: "consensus override changed; the node restarts with it at a safe moment"; its node is node 1 (external), already on 7bd98cc4, nothing to restart |
| 2b switch, PC 2 | 11:40:56Z | ran 11:41:23Z, "restarting the node with the new consensus parameters", igneumd started 11:41:25Z (pid 18732) on 7bd98cc4..., mining again 11:43:42Z, 113.0 MH/s at 11:44:42Z |
| 2b switch, PC 1 (last) | 11:42:54Z | ran 11:43:28Z, node restarted 11:43:29Z (pid 5556) on 7bd98cc4..., "waiting" 11:43:44 to 11:44:14Z, mining 11:44:44Z, 100.95 MH/s ramping at 11:45:14Z: its miners read 0 MH/s for about a minute after the node restart before coming back (the C43 class: the miner waits out the restarted node instead of resubscribing at once; the coordinator's note); the Ember Tune run 3 on PC 1 (ember-tune-pc1-3, 11:44:58Z) then took the box, after this restart, not under it |
| the laptop, Sam's Mac | off the air | they take 0.3.12 and the ten-field object through the manifest when they return; no 0.3.11 app was on the air to take the ten-field file before its binary (C39) |
## 7. The digest sweep (closed 11:45:20Z)
| Node | Binary | Digest | Since |
|---|---|---|---|
| the observer | `igneumd/2.1.0-83089544` (746a931f...) | 7bd98cc4... | 11:36:55Z |
| node 1 | the same | 7bd98cc4... | 11:37:07Z |
| the seed | 83089544 (4f142d51..., glibc 2.36 target) | 7bd98cc4... | 11:37:25Z |
| PC 2 | the installer's igneumd.exe f580b4aa... (the Mac cross-build) | 7bd98cc4... | 11:41:25Z |
| PC 1 | the same | 7bd98cc4... | 11:43:29Z |
| the Mac | attached to node 1 | node 1's | 11:37:07Z |
| the laptop, Sam's Mac | 0.3.10 / 0.3.9 | pending | off the air |
Tip 183,154 at 11:45:20Z, no refusals on the hand nodes after the switch; the fresh-record rule arms at DAA 198,000 (about 15:55Z at 0.98 DAA/s).
The fleet during the window: PC 2 and PC 1 mined on 0139ab9d while the hand nodes and the seed were on 7bd98cc4 (11:37 to 11:41Z); each
rejoined at its switch; the Mac's miner paused throughout on the project lead's order.
## 8. CI
| Run | On | Result |
|---|---|---|
| `ci` 37435705568 | ebea8b6 (the branch push, 08:22:40Z) | green (pow tests and census build, simulators, site build + link check + identity, the PowerShell 5.1 parse job) |
| `windows-ci` 37435975425 | ebea8b6 | green 08:30:58Z (the parse job 08:25:16 to 08:25:56Z; engine, window host, payload, installer, smoke run 08:26:00 to 08:30:58Z); superseded by the run below (no telemetry helper in its payload) |
| `ci` 37436904569 | 6532adf (the branch push, 08:33Z) | (pending) |
| `windows-ci` 37436904041 | 6532adf | green 08:39:20Z; superseded by the run below (the node changed) |
| `ci` 37436904569 | 6532adf | green |
| `windows-ci` 37438673235 | 062c3f8 (`gh workflow run windows.yml --ref release-0.3.12`, 08:49:12Z, after the second inputs push) | green 08:54:27Z (the parse job 08:49:18 to 08:49:59Z; engine, window host, payload, installer, smoke run 08:50:02 to 08:54:27Z against the 83089544 inputs); fetched 09:03:17Z with `OTA_SKIP=1 CONSOLE_SKIP=1 packaging/windows/fetch-ci-artifacts.sh 37438673235` into the downloads folder, NOT deployed |
| `ci` 37438674529 | 062c3f8 | FAILED in one step, `prover-socket-check.sh`: `tools/proving-v1/pc2-agg-cost.ps1` (in through the proving-v1 merge) ends its root prover with `pkill -x sp1-gpu-server`, and the check wants `pkill -f`; the line now reads `pkill -f ... ; rm -f /tmp/sp1-cuda-*.sock` (one playbook line, nothing the app or the packaging reads; `git diff 062c3f8 <fix> -- app packaging proto-cuda proto-opencl proto-metal vendor` is empty, so the Windows artefacts of 37438673235 stand, as 0.3.11's did across 3b0262f and 2a62735); the rerun is the row below |
| `ci` 37440456687 | 37b6a7f (the one-line playbook fix) | green 09:07Z |
| `ci` 37440559793 | 88df58e (master d3b64cb merged in: the morning summary, docs only; the release tip) | green 09:08:08Z. The 0.3.12 CI verdict is therefore run 37440559793 on 88df58e; the Windows build is run 37438673235 on 062c3f8, the same app, packaging and node sources (`git diff 062c3f8 88df58e -- app packaging proto-cuda proto-opencl proto-metal vendor igneum-pow` is empty) |
The ship state file `~/.cache/igneum/ship/0.3.12.json` carries `sha` = 062c3f8 (the Windows-build commit, which the ci step looks up by commit; the tree is 88df58e, docs and one playbook line later, as 0.3.11's was two docs commits past its build commit),
`forkCommit` 89dfcb95, `bumpedAt` before the DMG's mtime (so the dmg step reads "already"), and `runId` once the Windows run is green.
## 9. Owed to the next cut (0.3.13)
| Item | What |
|---|---|
| C43, the miner's dead gRPC channel | no commit exists; `igneum-miner mine grpc://` must re-subscribe after its node restarts (the 0.3.11 finding: 11 min of 26 MH/s burned on the Mac); until then every hand restart of node 1 is followed by `restart --what miners` |
| prover-floor's server | its packaging row is 0.3.13 (the coordinator's word) |
| explorer d7e797c, pool-v0, rig-install, ota-k2, the ledger forks | left out on the coordinator's word |
| the Windows workers' reproducibility | mingw PE builds differ byte-for-byte between builds of the same sources (12bfaa27 vs 2b3b8c92 today); a `-Wl,--no-insert-timestamp` (or `SOURCE_DATE_EPOCH`) in `build-windows.sh` would make G5's one-commit rule checkable by hash |
| the site build's non-idempotence and the pre-push flip | unchanged from 0.3.10 and 0.3.11 |

View file

@ -164,7 +164,7 @@ Every emitted instruction satisfies: `rot` in 1..31, `mask` in {1, 2, 4, 8, 16},
### 1.4.5 Encoding
A program is transmitted as the seed bytes, never as instructions. A node hands a miner the pack it emits itself (`igneum-pow/src/emit.rs`: `kernel.cu`, `kernel.cl`, `program.metal`, `program.h`, `program.json`, `memhard.h`, `vectors.*`, the three `*_bound` kernels), and a miner MAY regenerate everything from the seed bytes by the procedure of 1.4.6. `program.json` (format `igneum-program-pack-3`) is the interchange form; its field names are those of `Instr` in `generator.rs`, and it carries `generator` (2), `attempt`, `program_id` and `seed_bytes`. `program.h` carries the same as `IGNEUM_GENERATOR`, `IGNEUM_PROGRAM_ATTEMPT`, `IGNEUM_PROGRAM_ID` and `IGNEUM_SEED_BYTES_HEX`. An implementation MUST refuse a pack whose generator version is not its own.
A program is transmitted as the seed bytes, never as instructions. A node hands a miner the pack it emits itself (`igneum-pow/src/emit.rs`: `kernel.cu`, `kernel.cl`, `program.metal`, `program.h`, `program.json`, `memhard.h`, `vectors.*`, the three `*_bound` kernels), and a miner MAY regenerate everything from the seed bytes by the procedure of 1.4.6. `program.json` (format `igneum-program-pack-3`) is the interchange form; its field names are those of `Instr` in `generator.rs`, and it carries `generator` (2), `attempt`, `program_id` and `seed_bytes`. `program.h` carries the same as `IGNEUM_GENERATOR`, `IGNEUM_PROGRAM_ATTEMPT`, `IGNEUM_PROGRAM_ID` and `IGNEUM_SEED_BYTES_HEX`. An implementation MUST refuse a pack whose generator version is not its own. Program class v3 (Counter ASIC 2.0, 5 October 2026, activated by the height switch `program_class_v3_activation_daa` from the first epoch whose start score is at or above it, `docs/plans/counter-asic-2-rollout.md`) writes `generator` 3, and every pack of it carries `IGNEUM_PROGRAM_CLASS` (`v3`) and `IGNEUM_ERA_SEED_HEX` (the 32-byte era seed of section 1.13.1, or its devnet stand-in) beside `IGNEUM_GENERATOR`; the serve protocol's `prepare` and `job` lines carry `class=v3 era=<hex>` for v3 epochs and nothing for v2 ones. A worker MUST refuse a pack whose class or era seed does not match the line it was prepared for (`igneum-pow/src/packcheck.rs`, `verify_pack_dir_chain`; `proto-cuda/nvrtc/packfile.h`), and a pack of a generator other than 2 or 3.
### 1.4.6 Program acceptance
@ -178,7 +178,7 @@ Implemented (`igneum-pow/src/accept.rs`, `proto-metal/main.swift`; ledger M6 Fix
Attempts. Attempt 0 of a program seed `b` (the 32-byte epoch seed, or the UTF-8 of a seed string) is the candidate drawn from `seed_words_from_bytes(b)`. If it fails, attempt `k = 1, 2, ...` is drawn from `seed_words_from_bytes(b || k_le32)`; the first accepted candidate is the program of the epoch. Measured rejection rate under this generator: 5.14 percent over 100,000 seeds (census section 7) and the 20,000-seed confirmation of `docs/bench-log.md` (4 October 2026), so the probability that 32 consecutive candidates fail is below 2^-136, and an implementation MAY treat 32 consecutive failures as a consensus fault (`MAX_ATTEMPTS`).
Program id. `FNV-1a-64("igneum-program/" || generator_le32 || seed words as little-endian bytes || attempt_le32)` with `generator = 2`, written into every pack. Two implementations that agree on the id agree on the generator version, the seed words and the attempt.
Program id. `FNV-1a-64("igneum-program/" || generator_le32 || seed words as little-endian bytes || attempt_le32)` with `generator = 2` under class v2 and `generator = 3` under class v3, written into every pack. Two implementations that agree on the id agree on the generator version, the seed words and the attempt.
Why the closed form: the test is then a pure function of the program (no cache, no day), costs 1.3 to 3.4 ms on one core, and the census checked on 100,000 programs that its verdict agrees with the memory-hard dataset's on all but 39 threshold-edge cases (section 7.3). What the three parts catch: (a) the empty-list fallback of 1.4.3; (b) registers that saturate to all ones (2.4 percent of candidates); (c) zero-absorbing register sets, lane-constant load sites, output bias and value-level address repeats (2.1 percent). Not in the rule, and why: a contraction as the last write (80 percent of programs) and the `or` count are too common and (c) already catches the cases that matter; the load critical path is a hash-rate question, not a weakness.
@ -207,7 +207,7 @@ Implemented for the prototype size; Designed for genesis.
A load reads one 4-byte word at `src AND MASK`. Every load in every emitted kernel has exactly this form; the static check in `TESTS.md` section 5 is part of conformance (section 1.15). Because item values do not depend on the dataset size (section 1.8.5), the 1 GiB vectors remain valid for words below 2^28 at any larger size.
Growth beyond genesis is in section 1.13.
Growth beyond genesis is in section 1.13; under program class v3 the cache follows the dataset's doublings (1.13.3) and the verifier holds 256 MiB, then 512 MiB from year 4 and 1 GiB from year 12.
## 1.6 Register initialisation
@ -319,7 +319,21 @@ s = M_8(s)
item(t) = s
```
Eight dependent cache reads (`ITEM_ROUNDS = 8`, prototype value): the address of read `r` depends on every earlier read. Nine mixer applications. `dataset[w] = item(w >> 4)[w AND 15]`. A dataset of 2^D words is the prefix of items `0 .. 2^(D-4) - 1`, so an item has the same value at every dataset size.
Eight dependent cache reads (`ITEM_ROUNDS = 8`, prototype value): the address of read `r` depends on every earlier read. Nine mixer applications under program class v2.
Program class v3 (Counter ASIC 2.0, decided 5 October 2026, delegated; the project lead confirms for the public testnet genesis) applies the mixer `m = 8` times per round with distinct round keys (`LoadClass::mixer_mult`; `docs/plans/mixer-x4.md` section 2), the eight dependent reads unchanged:
```
for r in 0..7:
for j in 0..m-1:
s = M(s, rk = (r * m + j + 1) * 0x9E3779B9)
a = s[0] AND (2^(C - 4) - 1) cache line index, 2^(C - 4) lines of a 2^C-word cache (C from 1.13.3)
s[i] = s[i] XOR cache[line a][i] for i in 0..15
for j in 0..m-1:
s = M(s, rk = (8 * m + j + 1) * 0x9E3779B9)
```
Under `m = 1` the keys are `(r + 1) * 0x9E3779B9` and `9 * 0x9E3779B9`, the class v2 text exactly; the `9 m` keys are the first `9 m` values of the sequence `k * 0x9E3779B9`, all distinct. Why `m = 8`: the recompute attacker's cost is operations per item (`docs/analysis/m16-recompute-attacker-2026-10-05.md`); the honest miner pays the mixer once a day in the dataset build, which stays latency-bound (RTX 5090 23 to 25 ms, RX 9070 XT 72 to 77 ms, M5 Max 21 ms at `m` = 1, 4 and 8, measured 5 October 2026); the verifier pays `m` per item it derives: 0.61 ms per warp at `m = 1`, 1.24 at 4, 2.08 at 8 on one loaded M5 Max core (measured 5 October 2026, `docs/plans/mixer-x4.md` 6.4a), inside the 10 ms gate. The on-die-cache recompute chip's gain against the RTX 5090 falls from 2.4x (`m = 1`) to 0.92x with a 3x fixed-function factor at `m = 8` (`docs/analysis/chip-model-v3.md`, approximate factor). Under class v3 an item's value also depends on `C` through the line mask, so the items change on the day the cache doubles (1.13.3); the emitted `mh_item` carries the `m` loop only for `m > 1`, so every class v2 pack keeps its text. Class v3 also draws the dataset layout and the load windows per era (`docs/plans/era-layout.md`; the strided windowed load address and the interleaved mapping `mh_addr`, one text form in the three dialects). `dataset[w] = item(w >> 4)[w AND 15]`. A dataset of 2^D words is the prefix of items `0 .. 2^(D-4) - 1`, so an item has the same value at every dataset size.
What the construction buys (Measured, `MEMHARD.md` section 2.2, M5 Max, seed igneum-genesis, 1 GiB):
@ -396,11 +410,11 @@ All times are DAA seconds since genesis (section 0.6). At 1 block per second one
| Clock | Length | What changes | Label |
|---|---|---|---|
| Epoch | 3,600 DAA s | The program: new seed words from the VDF of section 4, new kernel | Designed (design document, "Always evolving, on three clocks"); the epoch length is a prototype value, to be fixed at gate 2 by the difficulty-tracking measurement (fork map c1: the hash rate steps by program, 35 to 48 Mhash/s across seeds on the M5 Max, so the DAA window must track within an epoch) |
| Epoch | `epoch_len(d)` DAA s, base 3,600; the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200; set by 90% miner signal at a day boundary (sections 1.13.1 and 5.7) | The program: new seed words from the VDF of section 4, new kernel | Designed (Counter ASIC 2.0 layer 9, 5 October 2026, `docs/plans/epoch-length.md`); 3,600 stays the value on every network until a signal moves it, and stays the prototype value to be fixed at gate 2 by the difficulty-tracking measurement (fork map c1: the hash rate steps by program, 35 to 48 Mhash/s across seeds on the M5 Max, so the DAA window must track within an epoch) |
| Day | 86,400 DAA s | The day key, hence the cache and the dataset | Designed |
| Era | 15,552,000 DAA s (180 days) | Era parameters and one instruction-family unlock, section 1.13 | Designed; the length is a prototype value (the design says "every 6 months") |
Epoch `e` covers DAA scores `[3,600 e, 3,600 (e + 1))`. The epoch of a block is the epoch of its own DAA score, so "which program was this block mined under" is a function of the header alone once the seed is known. The program for epoch `e` is `generate_from_seed_bytes(program_seed_e)`: attempt 0 is drawn from `S_e = seed_words_from_bytes(program_seed_e)`, and a rejected attempt is replaced as 1.4.6 says; `program_seed_e` is the 32-byte VDF output of section 4.3.
Epoch `(d, e)` covers DAA scores `[86,400 d + L e, 86,400 d + L (e + 1))` with `L = epoch_len(d)` and `e` in `0 .. 86,400 / L`; every ladder step divides 86,400, so day boundaries are epoch boundaries, and the epoch is identified by its start score `s = 86,400 d + L e`. At the base `L = 3,600` this is `[3,600 e, 3,600 (e + 1))` and nothing below differs from the earlier text. The epoch of a block is the epoch of its own DAA score, and `epoch_len(d)` is a function of the blue blocks of the signalling window that closed at least 2 days before day `d` (section 1.13.1), which are in the header's past, so "which program was this block mined under" is a function of the header alone once the seed is known. `T_epoch` and the 1,200-s lead of section 4.3 are genesis constants and do not follow `epoch_len`: the program of every epoch is known 600 s before it starts on the reference core at every length. The program for epoch `e` is `generate_from_seed_bytes(program_seed_e)`: attempt 0 is drawn from `S_e = seed_words_from_bytes(program_seed_e)`, and a rejected attempt is replaced as 1.4.6 says; `program_seed_e` is the 32-byte VDF output of section 4.3.
Implementation note (devnet, 3 October 2026, `docs/fork-divergence.md` "Epoch seed"): until the VDF of section 4 is in the node, `program_seed_e` is the hash of the last selected-chain block whose DAA score is below `3,600 e - 600`. The 600-DAA-score lead stands in for section 4.3's 20-minute lead: the program of epoch `e` is knowable about 10 minutes before it starts, every block template reports it (`pow_epoch.next_epoch_seed`), and a GPU worker compiles it in the background and swaps at the boundary with no pause (serve protocol `prepare`, `proto-metal/main.swift`, `proto-cuda/host.cu`, `proto-opencl/host.c`). Measured across boundaries on a short-epoch test network in `docs/bench-log.md` (hot-swap entry). The program schedule is a protocol constant; a miner that cannot compile ahead sees the same seed at the same time as everyone else, only later.
@ -422,12 +436,31 @@ The era seed `E_n` is the 32-byte output of the 1-hour VDF of section 4.4. One S
| Load count | 16 of 64 | not drawn | fixed, so every era is equally memory-bound |
| Output fold rotations | (7, 14, 21), (9, 18, 27) | each `1 + below(31)` | 1..31 |
| Mixer round count | 8 | not drawn | fixed, so the verify budget holds |
| Mixer applications per round `mixer_mult` | 4 (class v3; 1 under class v2) | not drawn | fixed at genesis (Counter ASIC 2.0, 5 October 2026: the M16 recompute chip's only measured lever; section 1.8.5 carries the form; `docs/plans/mixer-x4.md`) |
| Epoch length `epoch_len` | 3,600 DAA s | one draw of the era stream consumed and not used (the value is set by signal) | the ladder 600, 900, 1,200, 1,800, 2,400, 3,600, 4,800, 7,200 (layer 9, `docs/plans/epoch-length.md`) |
`epoch_len` is the one era-table parameter set by miners rather than by the draw: 90% of blue blocks over a 7-day window carrying the same ladder index (3 bits of the header version, encoding Open in section 5.8) sets that length from the first day boundary at least 2 days after the window closes (section 5.7). It is not a code upgrade: the rule, the ladder and the window are genesis constants, and the chain carries no release. The era stream consumes its draw so that a future draw of this parameter changes no other parameter's value. The threat it answers is a per-program hard datapath (an FPGA fleet: 42 to 160 minutes per compile on a mid-size part, PRflow, FPT 2019, hours on large parts; at 600 s nothing it compiles ever runs); it does not answer a programmable chip, which the other layers answer. The floor 600 is set by the slowest compile-ahead measured (the Metal variant race, 38 s on the M5 Max, 6.3% of a 600-s epoch and inside the 600-s seed window; `docs/plans/epoch-length.md` section 6).
The table layout and the working-set window (Counter ASIC 2.0 layers 4 and 8, decided IN on 5 October 2026, delegated: the six-era hash-rate spread is 1.3% on the RTX 5090, 3.2% on the RX 9070 XT and 0.8% on the M5 Max, under the 5% rule; `docs/plans/era-layout.md`) are drawn under program class v3 by a second stream `S` seeded with words 0 and 1 of `seed_words_from_bytes("igneum-era/" || E_n)` (the index is not in the preimage: `E_n` commits to `n` through the VDF input), seven draws in this order whether or not a value is used:
1. `W = allowed[below(|allowed|)]`: the width in words of every dataset load of the era, from the genesis-fixed set `allowed`; the set is `{1}` (4 bytes, the read-width decision of 5 October 2026), so the draw is consumed and the width pinned.
2. `M = low32(next()) OR 1`: the stride multiplier, odd, so `x -> x * M` is a bijection.
3. `R = 1 + below(31)`: the stride rotation.
4. to 7. `r_i = next()` for `i` in 0..3: the interleave draws. With `b = log2(W)` and `free = 4 - b`, `c = [b, ..., 15]`; for `i` in `0..free`: `j = i + (r_i mod (16 - b - i))`, swap `c[i]` and `c[j]`; the interleave is `pos = [0, ..., b - 1] ++ sort(c[0..free])`, four ascending bit positions below 16.
The era parameters are `(W, M, R, pos)`. Dataset mapping under class v3: word `w` holds word `j(w)` of item `t(w)`, where bit `i` of `j(w)` is bit `pos[i]` of `w` and `t(w)` is `w` with bits `pos[0..3]` removed; with `pos = [0, 1, 2, 3]` this is `dataset[w] = item(w >> 4)[w AND 15]` byte for byte; an item keeps its value at every dataset size of at least 2^16 words, and the `W` words of one aligned load lie in one item, so the 4,096-item verifier bound of 1.11 holds. Load address under class v3, for a load site with window draws `(k_off, o)` and a dataset of `2^D` words: `k = min(k_off, D - 26)`, `y = rotl(x * M, R)`, `idx = ((y AND (MASK >> k)) OR ((o AND (2^k - 1)) << (D - k))) AND MASK` (uniform on the window, branch-free, three operations before the mask), one text form in Metal, CUDA and OpenCL. The window draws per instruction (layer 8), after the nine draws of 1.4.3: `k_off = below(3)` (the dataset, a half or a quarter) and `o = low32(next()) AND (2^k_off - 1)`, used only on a load slot, so a class v3 program takes 720 draws; the window never goes below 2^26 words (256 MiB, above the largest on-chip cache in the benchmark) nor above the dataset, and sixteen sites with drawn offsets cover the dataset with high probability (a windows-union census over 300 programs: the SRAM mirror a chip would need is the whole dataset in every hour). The acceptance rule of 1.4.6 is unchanged in its tests and mirrors this address at its constant `D = 28`. Devnet stand-in for `E_n` until the VDF of 4.4 is in the node: era 0 the genesis block hash; era `n >= 1` the hash of the last selected-chain block whose DAA score is below `15,552,000 n - 7,200`. What the interleave buys and does not: a chip that hard-wires one layout reads the wrong 15 words with every word once the era draws another; a chip whose address decoder can permute its address lines pays nothing (stated in the plan). The stride is a bijection with no cryptanalysis yet (Open).
"Memory pattern" in the design document is read here as the item-address pattern (the cache line index word, `s[0]` in 1.8.5, and the XOR-all-sixteen rule); the proposal is to leave it fixed at era 0 and let the unlocked families change the kernel instead, because every change to the item derivation changes the verify time and must be re-measured.
### 1.13.2 Instruction-family reserve
At genesis the generator carries the eleven families of 1.4.1 live and a reserve list of further families in a fixed order. At the start of era `n >= 1`, reserve family `n` becomes live with weight `W_new` taken proportionally from the live non-load families. A family may enter the reserve only if it is integer-exact and has passed the cross-vendor conformance of section 1.15 on every vendor in the benchmark (Metal, CUDA, OpenCL on NVIDIA and AMD), with its own edge-case vectors, before genesis. Candidate families, all integer ALU operations present on Apple, NVIDIA and AMD: variable left shift and logical right shift by `src AND 31`; bit-field extract with an immediate offset and width; `andn` (`dst = dst AND NOT src`); byte permute of `dst` by an immediate selector; population count and count-leading-zeros folded into `dst` by add; a three-register select (`dst = bit of src2 ? src : dst`); a second shuffle form (`lane + delta mod 32`). The order and `W_new` are Open. A family that is not in the genesis reserve can only be added by the upgrade path of section 5.7.
At genesis the generator carries the eleven families of 1.4.1 live and a reserve list of further families in a fixed order. At the start of era `n >= 1`, reserve family `n` becomes live with weight `W_new` taken proportionally from the live non-load families. A family may enter the reserve only if it is integer-exact and has passed the cross-vendor conformance of section 1.15 on every vendor in the benchmark (Metal, CUDA, OpenCL on NVIDIA and AMD), with its own edge-case vectors, before genesis. Candidate families, all integer ALU operations present on Apple, NVIDIA and AMD: variable left shift and logical right shift by `src AND 31`; bit-field extract with an immediate offset and width; `andn` (`dst = dst AND NOT src`); byte permute of `dst` by an immediate selector; population count and count-leading-zeros folded into `dst` by add; a three-register select (`dst = bit of src2 ? src : dst`); a second shuffle form (`lane + delta mod 32`). The order and `W_new` are Open, except the first entry, decided 5 October 2026 (Counter ASIC 2.0 layer 7, delegated; the project lead confirms for the public testnet genesis; `docs/analysis/int8-matrix-family.md`):
> Reserve family R1, `mm8` (integer matrix). Semantics: section 2.2 of `docs/analysis/int8-matrix-family.md`, uint8 operands from `src` and `src2` in the m8n8k16 fragment layout, one int32 element of C per lane selected by the immediate `bit`, added into `dst` modulo 2^32. Weight at unlock `W_new = 4` points, taken proportionally from the ten live non-load families (the load weight and count are untouched). Edge vectors, each a hand-built unit run on every vendor: all bytes 0xFF in A and B (C = 1,040,400 everywhere); all bytes 0x80 (C = 262,144); A all zero (C = 0); `dst` = 0xFFFFFFFF with a nonzero C (the wrap); alternating 0x00 and 0xFF by lane; `bit` = 0 and 1 on the same fragments. Unlock: at the start of era n = 4 (DAA 62,208,000), or earlier by the 90% signalling path of section 5.7; never by a release. Native paths: PTX `mma.sync` `.u8` (sm_75+), AMD WMMA `i32_16x16x16_iu8` (RDNA 3 and 4), Metal 4 `mpp::tensor_ops::matmul2d` (`uchar x uchar -> int`); the per-lane `dot4` form is emulation on Apple (1.6x per op unsigned, measured 5 October 2026) and is not the reserved form.
A vendor that can only emulate. A family enters the reserve when it is bit-exact on every vendor of 1.15. A vendor that reaches the result only by emulation (no instruction or library path) does not block entry if the measured penalty of the emulation on that vendor, on the family's own probe (a dependent chain of the op against the same vendor's integer ALU chain), is at most 8x per op, AND the family's weight at unlock keeps the emulating vendor's hash-rate loss under 5% on the memory-hard hash, checked on the vendor's card with the family live. A family whose emulation exceeds either bound stays out of the reserve until the vendor ships a path.
A family that is not in the genesis reserve can only be added by the upgrade path of section 5.7.
### 1.13.3 Dataset growth
@ -437,7 +470,7 @@ Designed: 2 GiB at genesis plus 0.5 GiB per year (design document, "Which cards
N_d = floor((2 GiB + 0.5 GiB * (86,400 d / 31,536,000)) / 64 bytes)
```
evaluated in integers (bytes), with one year = 31,536,000 DAA seconds. The dataset grows by about 23 KiB per day and is recomputed with the day key. Two consequences are Open:
evaluated in integers (bytes), with one year = 31,536,000 DAA seconds. The dataset grows by about 23 KiB per day on this average and is recomputed with the day key. Decided 5 October 2026 (Counter ASIC 2.0 layer 6, delegated; the project lead confirms for the public testnet genesis): the cache grows with the dataset, doubling when the dataset doubles: `cache_log2_words(d) = 26 + growth_doublings(d)`, `growth_doublings(d) = floor(log2(1 + d / 1,460))` for day `d` since genesis (doublings at years 4, 12, 28 and 60), so the cache is 256 MiB at genesis, 512 MiB from year 4, 1 GiB from year 12; the verifier's one-core fill is 0.2, 0.4 and 0.8 s at those steps (0.2 s per 256 MiB, section 1.12), under 1 s at every step of the schedule. The dataset steps to the next power of two on the same doublings (option (b) below, recommended to the project lead with the card-lifetime consequences in `docs/analysis/card-lifetime-2026-10-05.md`: a 4 GB card mines to year 4, an 8 GB card to year 12, a 12 GB card to year 28 with the cache freed after the daily build). Why the cache grows at all: an SRAM mirror of a flat 256 MiB cache is about 128 mm^2 and $46 of silicon at N5 by shipped cache-die density (AMD V-Cache, 64 MB on 41 mm^2 at 7 nm; `docs/analysis/sram-mirror.md`), so the cache size never prices a chip out; its job is to stay above any GPU's on-die cache (96 MB on the RTX 5090, 128 MB on GB202), which a flat 256 MiB loses within the decade. The GPU frees the cache after the daily dataset build; the hash never reads it. Two consequences remain Open:
- Index mapping. `src AND MASK` requires a power-of-two size. For a non-power-of-two `N_d` the proposed mapping is `idx = (src * N_words) >> 32` computed in 64 bits (a multiply-shift range reduction; uniform to within 2^-32, branch-free, integer only). At `N_words = 2^28` this gives `src >> 4`, not `src AND MASK`, so adopting it changes the 1 GiB vectors; gate 1 chooses between (a) the multiply-shift mapping with new vectors, or (b) power-of-two sizes only, growing in steps (2 GiB, 4 GiB) on the same schedule's average, which keeps `AND MASK` and means a 4 GiB card lasts until the 4 GiB step instead of fading.
- The item index `t` is 32 bits, so the construction as written tops out at 2^32 items = 256 GiB, which the schedule reaches after 508 years. No action needed.
@ -509,4 +542,6 @@ Pack `proto-cuda/packs/igneum-devnet-v4-epoch0/`: epoch seed bytes `edc4fa844da9
Header-bound vectors (section 1.6 rule, seed `igneum-genesis`, day `2026-10-03`): `igneum-pow/README.md`, eight values, for example H = 32 zero bytes and nonce 0 give `746c567b090acf6a`.
Program class v3 vectors (5 October 2026, `proto-cuda/packs-ca2-mixer/` and `proto-cuda/packs-ca2-era/`; generator 3, mixer x8, the cache growth rule, the era draw inside the class): pack `mx8-genesis` (seed `igneum-genesis`, no era, program id `e323b9dcaf283a6f`, batch fingerprint `7c28cfb06c5c65a9`) and pack `mx8-devnet-epoch0` (the devnet epoch seed and day of 1.17 with the era stand-in E_0 = the devnet genesis hash inside the class, program id `73bcbfe8ccf988f1`, unit 0 lane 0 `d424577fce4a7a60`, batch fingerprint `90f794dd556f7a3b` over 2^24 outputs at base nonce 0), reproduced by the Rust interpreter, Metal and Apple OpenCL on 5 October 2026 (3/3 standalone, 3/3 in batch, 96 of 96 lanes each); the six era packs `era-0` to `era-5` (the same epoch seed and day, era test seeds 0 to 5, program id `73bcbfe8ccf988f1`; 2^24 fingerprints `8e8e070db4eea52d`, `891c01b8563bb47e`, `e54279fed2831b5d`, `77e0ba8abbd0ae62`, `d898d8f4f2e7684b`, `a6927db380f7efb2`). All seven fingerprints are equal on the RTX 5090 (CUDA/NVRTC), the RX 9070 XT (AMD OpenCL) and the M5 Max (Metal), self-test PASS on every pack, and 1,024 random nonces per card on `era-0` and `mx8-devnet-epoch0` re-hash to the same value on the Rust verifier (1,024 of 1,024 each): job `run-ca2-era-pc1b-20261005`, 5 October 2026, `docs/bench-log.md`. The class v2 vectors above stand unchanged (the v2 exports are byte-identical on the class v3 crate, `igneum-pow/tests/packs.rs`).
Cache, dataset and mixer vectors: section 1.8.4 and 1.8.5. Seed words: section 1.3.1. Generator: section 1.4.3. Acceptance: section 1.4.6. Batch fingerprints: section 1.15 item 5.

View file

@ -32,7 +32,7 @@ All in public, on the hashrate charts. 51% never reaches 2/3 while honest miners
- **C1.** Checkpoint i is the selected-chain block at blue score 30 i. It is determined once the virtual's blue score reaches 30 i + d. d = 60 at 1 block per second is a placeholder (ledger F7): the gate 3 devnet records the reorg-depth distribution and sets d so that a vote split at one index is rare and self-heals at the next. d scales with block rate. Re-determination (rule of 4 October 2026, night, ledger F24): while index i is not locked, a node whose selected chain moves past C_i (a reorg deeper than d) determines index i again on its new chain; its own votes for the old block stand (a key never signs two blocks at one index) and a certificate the network formed over the new block, received meanwhile and held pending, is then verified. A locked index is never re-determined (3.11.4): fork choice keeps the chain through its block, and a certificate for another block there is a conflict (C4). A lock lands about 90 to 120 s after a transaction (Designed; simulated lock latency after the checkpoint block is median 2.5 s, p99 4.6 s at a 2-s inter-region delay, `sim/results_v2.md` A).
- **C2.** A vote is a BLS signature over `(chain_id, i, hash(C_i))` under a fixed domain-separation tag. Votes gossip as their own message type.
- **C3.** A lock certificate for index i is an aggregate BLS signature over one checkpoint block hash with a bitmap of signers, whose signed weight meets Q3. Every block carries the highest certificate its producer knows. A block whose selected chain does not pass through every certified checkpoint in its past is invalid (section 2.4).
- **C4.** A node holding a LOCK at index i rejects any other certificate for index i and publishes the pair as evidence (section 3.6). A certificate over a block that is not the node's own determination at an index it has not locked is not a conflict: the chain may still move to that block (C1 re-determination, ledger F24), so the node keeps it pending, bounded, until it does or the index is left behind. A certificate naming a block that cannot be index i's checkpoint on any chain (its blue score is under 30 i, or its selected parent's is not) is refused outright.
- **C4.** A node holding a LOCK at index i rejects any other certificate for index i and publishes the pair as evidence (section 3.6). A certificate over a block that is not the node's own determination at an index it has not locked is not a conflict: the node verifies it at that block and, when it is valid there, locks the index on it and moves its chain (3.5, the certificate-driven reorg, rule of 5 October 2026 night, ledger C4). A block the node does not have yet is kept pending, bounded, and tried again as the DAG arrives (ledger F24). A certificate naming a block that cannot be index i's checkpoint on any chain (its blue score is under 30 i, or its selected parent's is not) is refused outright.
- **C5.** No certificate may form in the chain's first 3,600 DAA seconds (design document). See 3.8 for the proposed first-month rule.
## 3.3 Quorum
@ -114,6 +114,8 @@ The honest level is therefore the number of indices at which k actually voted an
- **F4.** There is no hidden-block penalty in consensus. It was removed in review round 2 because it breaks DAG determinism and amplifies eclipse attacks. First-seen MAY break ties in a node's own block template only.
- **F5.** A node started with a configured trusted certificate follows it. A node started cold selects the DAG with the most accumulated blue work, then follows certificates found in it. A private DAG that out-works the public one over the window is a public 51% event lasting weeks.
**Certificate-driven reorg (rule of 5 October 2026, night; ledger C4; design document Finality v2, Fork choice items 1 to 4).** A node that holds a valid certificate for a checkpoint block that is not on its selected chain MUST move its virtual to the heaviest tip through that block. Valid means: the block is index i's checkpoint on its own chain (C1), it lies on the chain through every lock the node holds, the certificate verifies against the canonical voter list at that block (C3), and its signers meet Q3 and, under rule v3, Q5 there, every input a function of the block's own past. The node records the lock at i on that block, replaces its own record there, and determines its unlocked records again on the new chain (C1 re-determination). Merge depth does not bound this move. Finality outranks merge depth by design (item 2 above: a certified checkpoint removes other tips from candidacy, blue work decides only among candidates), and the certified chain's blocks are valid under their own merge-depth roots. The depth-based finality point does bound it, as F1 already implies: a certified block that is not in the future of the node's depth-based finality point is beyond what any rule can follow (section 2's pruning safety) and needs the operator's trusted certificate (F5). A certificate for a chain that misses a lock the node holds, at any index, is a conflict under 3.11.4: the lock stands and the pair is reported. While a node holds locks, its own determination at a higher index locks only on the chain through them. Before this rule the node held such a certificate pending a reorg that GHOSTDAG alone never produced, and a 96-s honest partition with no attacker ended in a permanent finality fork (bench-log "FUD ledger sweep round 6", C4; the fix and its measurement under "the C4 fix", 5 October 2026 night).
What a node does when it holds two valid certificates at one index after a partition heals is not modelled and not defined (`sim/results_v2.md`, "cannot tell us"; O-3.6). C4 says it publishes the pair. The proposal for gate 3: both certificates are evidence against every key that signed both; the node re-evaluates both against Q3 with those keys' weight struck, and if exactly one still locks it follows that one; if neither or both still lock, F2 decides among the two checkpoint blocks' descendants and the index is treated as uncertified.
## 3.6 Equivocation evidence
@ -170,7 +172,7 @@ Status of this section: Implemented in `vendor/igneum-node` (reading guide in `d
| C1 | Checkpoint i is the lowest selected-chain block with blue score at least 30 i (blue scores along the chain can skip values), determined when the sink's blue score reaches 30 i + d, d = 20 on devnet. Re-determination (branch `fud-consensus`, 4 October 2026 night, ledger F24): after every virtual change, every unlocked record whose block is no longer a chain ancestor of the sink is determined again on the new chain (`on_virtual_changed`, "re-determined" log line); the certificate held over the old block is dropped, the fold clock restarts, and the certificates kept pending over the new block (`pending_certificates`, at most 4 per index, indices up to 64 ahead of the next determination) are verified. A locked record is never revisited. Unit test `reorg_past_an_unlocked_checkpoint_re_determines_it_and_verifies_the_pending_certificate` (a 6-block side chain's certificate is pending with no conflict, the 15-block side chain overtakes, index 13 is re-determined and locks from it; a block with the wrong blue score is refused) and `a_locked_checkpoint_pins_the_chain_and_a_certificate_against_it_conflicts` (a side chain twice as long does not become the sink past a lock, the certificate against the lock is the one conflict) | d = 20 is below the placeholder 60; the devnet reorg-depth distribution that sets d has not been recorded. Measured in `docs/bench-log.md`, "round-4 consensus items" (reorg run) |
| C2 | BLS signature over `"igneum-vote-v1/" \|\| chain_id \|\| 0 \|\| index \|\| hash(C_i)` under `IGNEUM_VOTE_V1_BLS12381G2_XMD:SHA-256_SSWU_RO_NUL_`; the chain id is the prefixed network name (`igneum-devnet`, `igneum-devnet-7`); votes are p2p message 70 and ride in the coinbase extra data of every block | |
| C3 | Certificate = index, checkpoint, voter count, signer bitmap over the canonical voter list (keys above dust and not stripped, sorted by key hash), aggregate signature, aggregator key hash and sortition proof. Every template carries the certificates not yet in its past | The validity rule (a block whose selected chain misses a certified checkpoint is invalid) is NOT enforced; only fork choice (F1, F2) is |
| C4 | A certificate at an index for a block other than the one LOCKED there is kept and logged as CONFLICTING (`conflicting_certificates`); at an unlocked index it is held pending (F24 above), not logged as a conflict | Not published as evidence. The rule is now fixed by 3.11 item 4 (the node keeps the certificate it verified first, never re-evaluates it, and reports the conflict); the node does not yet clear `finality_active` or expose `finality_conflict` when the pair appears. Until 4 October 2026 night a reorg deeper than d made the node log every certificate at the moved index as CONFLICTING (ledger F24) |
| C4 | A certificate at an index for a block other than the one LOCKED there is kept and logged as CONFLICTING (`conflicting_certificates`), as is one whose block is not on the chain through the node's nearest locks at any index; at an unlocked index over a block this node holds it goes through the certificate-driven reorg (`ingest_off_chain`, 5 October 2026 night, ledger C4: verified against `voters_at` of that block, Q3 and Q5 by `quorum_at` from the block's own past, then LOCKED there, "LOCKED by certificate" log line, and the virtual processor is asked to resolve again, `VirtualStateProcessingMessage::Resolve`); over a block this node does not have it is held pending (F24 above) and tried again on every virtual change | Not published as evidence. The rule is now fixed by 3.11 item 4 (the node keeps the certificate it verified first, never re-evaluates it, and reports the conflict); the node does not yet clear `finality_active` or expose `finality_conflict` when the pair appears. Until 4 October 2026 night a reorg deeper than d made the node log every certificate at the moved index as CONFLICTING (ledger F24) |
| C5, 3.8 | `min_daa` = `weight_window` (2,592,000 DAA s on mainnet, 7,200 on devnet; a unit test pins the equality). `evaluate` never locks, and `ingest_certificate` refuses a certificate from any source, while the checkpoint's DAA score is below `min_daa`; the node logs "finality not active, window filling, N of M" at every determination until the sink's DAA score reaches `min_daa` and reports the same through `getFinalityCheckpoints` (`finality_reason`, `window_filled_daa`, `window_full_daa`). Unit test `processes::finality::tests::no_certificate_while_the_window_is_filling`: one key holding 100% of the weight signs every checkpoint of a 150-block chain at a 60-DAA window; nothing certifies below DAA 60, a hand-built certificate at an early index is refused, every checkpoint from DAA 60 locks (fin-fixes, 4 October 2026) | Implemented on 3.8's recommendation ahead of the launch-month simulation (O-3.1), which is still not run; gate 3 can lower the gate but not remove it without reopening ledger F1. The sink's DAA score the report compares is the one the virtual processor last handed the manager, so a restarted node reports the window as filling until its first virtual resolution |
| Q1, Q2 | Presence window 20 indices on devnet (240 mainnet). Block reading: participation counts the indices in `[i - P, i - 1]` at which a vote by the key is carried by any block, blue or red, in the past of C_i; a key whose first block in the window is younger than P x 30 DAA seconds counts the full window; every template carries up to 48 votes not already in its past, certificates and evidence first | The per-block vote bound (48) is the devnet value of O-3.3. Participation is credited for any vote by the key at the index, whatever block it names; 3.11.1 requires the vote to name the checkpoint on the crediting chain, else a key can stay in the active denominator by voting for blocks of its own and never add to a certificate (O-3.19) |
| Q3 | Integer tests: `3 x signed x P >= 2 x active_num` (active_num = sum of weight x participation count) and `3 x signed >= 2 x total` (was `30 x signed >= 17 x total` until 4 October 2026; `FinalityParams::FLOOR_NUM / FLOOR_DEN` = 2/3 on branch `devnet-v4`, with `quorum_met`, `floor_met` and `locks` as pure functions), both inclusive, both at C_i; bans known at evaluation time are applied to the voter list. Unit test `floor_is_two_thirds_of_total_and_inclusive`: 4 of 6 locks, 3 of 6 does not, 67 of 100 locks, 66 does not, the total test implies the active test for every participation. Measured on the three-node, six-voter network of `docs/bench-log.md`, "finality floor 2/3" (4 October 2026): no lock on either side of a 3/3 split, the 4 side of a 4/2 split locks at exactly two thirds | |
@ -178,13 +180,13 @@ Status of this section: Implemented in `vendor/igneum-node` (reading guide in `d
| Q5 | Rule v3 (same branch and switch): `frozen_table` finds the highest locked index below i whose block is an ancestor of C_i (`state.locks`, reachability), takes `voters_at` of that block with the bans known now, and drops it when `daa(C_i) >= daa(C_f) + weight_window`; `evaluate` requires `floor_met(frozen_signed, frozen.total)` of the signers (and of a held certificate's signers) on top of Q3; a locked checkpoint is never downgraded. The LOCKED log line carries the frozen fraction and the frozen lock's index; a checkpoint that passes Q3 and fails Q5 logs "held by the frozen table" at debug. Unit test `frozen_table_holds_a_side_without_the_other_keys_for_one_window`: A at 60% and B at 40% lock together; B leaves; under v2 A locks alone within 30 DAA of B's last block, under v3 not before the last lock is one window old, and then it does | The reference is the node's own highest lock on the chain (not the certificate carried in C_i's past), so a node that has not seen the newest certificate tests against the previous lock's table, which in a connected network differs by 30 s of blocks |
| S1 | VRF output = SHA-256 of the voter's BLS signature over `"igneum-sortition-v1/" \|\| chain_id \|\| 0 \|\| index \|\| hash` under the sortition tag (unique per key and message, so the signature is the proof); eligible when `output x total_weight < 8 x weight x 2^64`, drawn by weight (W6, ledger F17, fin-fixes 4 October 2026): the expected number of aggregators is 8 by weight whatever the key count, a key with no weight never draws, a key holding 1/8 of total weight or more always does (so with 8 or fewer equal voters everyone is eligible). Unit test `sortition_is_by_weight_not_key_count`: 200 dust keys draw nothing, 6 real keys draw `sum min(1, 8 w / T)`, 16 equal keys draw 8.00, a key split into 10 or 200 parts draws what it drew whole | Was `output x voters < 8 x 2^64` (per key) until 4 October 2026; measured on the attack harness (`docs/bench-log.md`, "finality v2 attack harness" S2, then "finality fixes F17 and F1"). A key above 1/8 of total weight that splits itself gains seats (its single ticket was capped at 1); seats carry no reward and no power, since anyone MAY aggregate and Q3 is tested by weight |
| S2 | Not implemented (sub-user sortition above 8,192 voters) | |
| F1, F2 | In `resolve_virtual` the highest locked checkpoint that is in the future of the depth-based finality point and in the past of some body tip replaces the finality point: tips outside its future are not sink candidates | A lock that no body tip passes through is logged and ignored for that resolution |
| F1, F2 | In `resolve_virtual` the highest locked checkpoint that is in the future of the depth-based finality point and in the past of some body tip replaces the finality point: tips outside its future are not sink candidates. Since 5 October 2026 night (ledger C4) a lock can be a block off the node's selected chain, adopted from a certificate (`ingest_off_chain`), so this is the certificate-driven reorg: the sink search keeps only tips through it, whatever the blue work of the old chain and whatever the merge depth; `evaluate` locks the node's own determination only on the chain through its nearest locks (`off_lock_chain`) | A lock that no body tip passes through is logged and ignored for that resolution; a lock not in the future of the depth-based finality point (a certified chain deeper than the finality depth) is logged once and cannot be followed (F5) |
| F3 | Not implemented: the pruning point and `virtual_finality_point` ignore locks | Must land before any pruning network |
| F5 | Not implemented (trusted certificate at start) | |
| 3.6 | A second vote by one key at one index for another block is evidence, carried in blocks (`EvidenceRecord`: the two votes, the carriers with their DAA scores). Branch `fud-consensus` (4 October 2026 night, ledger F23): the ban at checkpoint C is computed from C's own past (`bans_at`): the key is stripped at C when a carrier lies in C's past and `daa(C) < daa(lowest carrier) + ban` (7,200 DAA seconds on devnet); detection over RPC or gossip only puts the evidence into this node's templates ("detected here: carried in this node's next block"). Evidence records are bounded (4,096, the oldest dropped; a record goes once its ban ended two windows below the sink or it was never carried within one ban of being seen; 16 carriers per record). Unit test `ban_is_decided_by_the_carrying_block_so_nodes_agree_on_every_voter_list`: three nodes on one chain, one takes the equivocating vote over RPC, two see it from the carrier block only; the voter list agrees on all three at every checkpoint, the key is a voter before the carrier and after the ban and nowhere in between, and the third node verifies the first two's certificates at every locked index | A carrier the reachability store no longer holds (pruned) counts as in the past of every checkpoint more than the merge depth younger than it. Until 4 October 2026 night the ban was stamped node-locally (ledger F23). Measured in `docs/bench-log.md`, "round-4 consensus items" (ban run) |
| 3.9 | `getFinalityCheckpoints` reports `finality_active` (the window is full and a lock exists within the last P indices), the latest lock, and since 4 October 2026 `finality_reason` (`active`, `window filling, N of M` with N the sink's DAA score capped at `min_daa` and M `min_daa`, or `paused`) with `window_filled_daa` and `window_full_daa`; the miner prints a `FINALITY` line whenever the reason changes | `last_certified` as a DAA score is not reported; the conflict reason of 3.11 item 4 (two certificates at one index) is not reported (O-3.17), so a conflict still reads as `active` or `paused` |
| 3.11 item 6 (seed source) | The devnet keys the hourly program on the header's own `daa_score` (`epoch_seed`, `docs/review/round-3-2026-10-03.md`, R3.26), not on a checkpoint block | The `seed_source` rule (section 4.3 with the uncertified fallback of 3.11 item 6) is not implemented; nothing on the devnet exercises a seed during a finality pause |
| 3.11 test table | The four-miner test network of the bench-log entry is the only measurement on a real DAG: 93 checkpoints, 0 conflicting certificates, one equivocation strip, one 12-checkpoint pause under the floor, one heal | d = 20, presence 20 indices and a 7,200-s window are devnet values; the measured pause and heal are at those values, not the mainnet ones |
| 3.11 test table | The four-miner test network of the bench-log entry is the only measurement on a real DAG: 93 checkpoints, 0 conflicting certificates, one equivocation strip, one 12-checkpoint pause under the floor, one heal; since 5 October 2026 night the fast-time harness row for the certificate-driven reorg (3.11.7) | d = 20, presence 20 indices and a 7,200-s window are devnet values; the measured pause and heal are at those values, not the mainnet ones |
Node state is one persisted blob (`DatabaseStorePrefixes::IgneumFinality`), written at most once a second; votes received over RPC but not yet carried by a block are lost on restart, votes in blocks are not.
@ -280,6 +282,7 @@ Each guarantee, the scenario that tests it, and the measured result. Bench-log c
| Acquired keys decay as the window moves (3.11.5) | results K, keys worth 20% and 40% bought, 30% hashrate, signing and silent, 30 days, five seeds | share follows b (1 - t/30) + 0.3 t/30 within 0.6 points at every sampled day in every seed; keys worth 20% rise to 30% on day 30 and never reach 1/3; keys worth 40% hold the veto from day 1 to day 19 or 20 (formula 20) and end at 30%; withholding its votes, the 40% buyer stalls 63,307 to 68,716 of 86,400 checkpoints in 30 days (the pause lasts until it has decayed below one third) and the 20% buyer 304 to 1,045; 0 conflicts (K, floor 2/3) |
| Signing stops while mining continues, 1, 6, 24 h | results J and L1 | 34% and above: every checkpoint stalled for the whole silence; 33%: 665 to 727 of 720 in 6 h; 32%: 35 to 221; 30%: 0 to 40; first lock after resume 0 min at every weight; 0 conflicts (J, L1) |
| Seeds during a pause (3.11.6) | devnet epoch boundary through a forced pause | not yet run (O-4.3 implementation) |
| A certificate over a chain the node is not on: the certificate-driven reorg (3.5) | fast-time harness `tools/finality-attacks/c4.mjs`, weight against work, 130-s split, 240-DAA window, rule v2 (the live devnet's) | the work-majority node fetched the certified chain, locked 12, 13 and 14 by certificate within 2 s of the first block, re-determined 11, and all three nodes ended on the certified chain with 0 conflicting certificates and 0 disagreeing locks; the same under rule v3 with a 140-s split: the certifying side locked 11 and 12 during the split, the work-majority node adopted 12 by certificate 3 s after the heal and every node ended on the certified chain, 0 conflicting, 0 disagreeing; the module-off control took the heavier chain (bench-log "the C4 fix", 5 October 2026 night) |
| Two certificates at one index: no lock withdrawn (3.11.4) | devnet with a forced double certificate | not yet run (O-3.17) |
| `T` under the block reading (3.11.3) | O-3.3 re-run | not yet run (O-3.18) |
| Participation credited only for the chain's checkpoint (3.11.1) | results C with an adversary voting for private blocks | not yet run (O-3.19) |

View file

@ -49,7 +49,7 @@ Designed. Epoch e is the DAA-score interval `[3,600 e, 3,600 (e + 1))` (section
1. **Seed checkpoint.** `C(e)` is the highest-index checkpoint (section 3, C1) whose checkpoint block has DAA score at most `3,600 e - 1,200`: the latest checkpoint at least 20 minutes of DAA time before the epoch starts. The checkpoint block hash is Kaspa's full header hash, which covers the nonce, as the grinding defence requires (`proto-vdf/README.md`: a hash that covers only the body would let a miner start the VDF while still searching nonces).
2. **Evaluation.** `input = hash(C(e))`, `T = T_epoch`, run 4.2. `program_seed_e` is the 32-byte output; `proof_e` is (T, y, pi).
3. **Program.** `S_e = seed_words_from_bytes(program_seed_e)` (section 1.3.1), program = `generate_from_words(S_e)` (section 1.4). The 1,200-s lead is 2x the reference evaluation time, so a core half as fast as the reference still finishes before the epoch (4.6).
3. **Program.** `S_e = seed_words_from_bytes(program_seed_e)` (section 1.3.1), program = `generate_from_words(S_e)` (section 1.4). The 1,200-s lead is 2x the reference evaluation time, so a core half as fast as the reference still finishes before the epoch (4.6). The lead and `T_epoch` are fixed whatever the epoch length of section 1.12: at the floor of 600 DAA s the checkpoint is two epochs back and the program is known one full epoch ahead; at the base it is known for the last sixth of the previous epoch (`docs/plans/epoch-length.md`, section 3).
4. **Header.** Every header carries `seed_source = hash(C(e))` for its own epoch (section 2.4). A header is valid under the lottery only if its `seed_source` is a block on its own selected chain at the blue score of checkpoint index `i(C(e))`, and its PoW verifies under the program derived from that block. The proof is not in the header: a node verifies `proof_e` once per epoch (4.5 ms) and caches `S_e`.
Determinism of step 1 is the point of the rule: which block is "the checkpoint at blue score 30 i" is a function of the header's own past, so two nodes validating the same header derive the same program, and a header mined under a reorged-away checkpoint names a block that is not on its chain and is invalid. Whether `C(e)` must be certified (section 3) or merely be the selected-chain block at that blue score in the header's past is Open (O-4.3): requiring certification couples mining to finality liveness (a stall longer than the lead would stop the program from being derivable), which the design document accepts ("an epoch cannot start without a valid proof") and this specification argues against, because the chain is meant to keep running on plain GHOSTDAG through a finality pause (section 3.7 item 2). The proposal: the selected-chain block at that blue score, certified or not, deep enough (1,200 DAA s plus d) that a reorg across it is a merge-depth-scale event.

View file

@ -36,6 +36,8 @@ Self-dealing: a developer who also mines the including block collects 80% plus 2
Designed. The 20% emission share (section 2.5) and the provers' part of the 80% tip share are paid per block as a fixed amount for that block, divided among the block's shards by consensus proving cost, so a stuffed block earns no more than an honest one. Shards are not claimed first-come and carry no bond: each shard is assigned by sortition to 8 eligible provers for a 10-s exclusive window, then open to anyone, and the first valid proof included in a block is paid (section 7.2, decided 3 October 2026, ledger P8, C9). The parameters 8 and 10 s are set on the phase 4 devnet (O-5.1). A withheld shard costs nothing to bond against because nothing waits on an assigned prover: an unproven block delays only its proof; execution and the 30-s lock do not wait for it (ledger P9). The bond, slashed on a bad or late proof, remains in the external job market (5.4), where a customer does wait; its size and timeout are Open (O-5.6).
Proving v1 (section 7.8, 5 October 2026, Implemented behind `proving_v1_activation_daa`): from the switch, `proving_v1_aggregator_share_bps` of a block's fixed amount (a tenth, the project lead's decision at 0.3.11) goes to the aggregator whose segment record attests the block, the rest to the shards as before; a block in a segment that stays unproven past `proving_v1_unproven_daa` pays no aggregator share.
## 5.4 External job market
Designed. At launch external proving jobs are paid on the customer's chain, in the customer's currency, to a payout contract keyed by miner address, because Igneum cannot yet see Ethereum; the customer chain's own bond and slashing apply (design document, "The first six months"). When the job market settles on Igneum, which needs the proof bridge (phase 2 consensus proof, ledger P4, E7), every job fee paid in IGN splits:

View file

@ -54,7 +54,7 @@ An item closes when its measurement is in `docs/bench-log.md` or its decision is
| O-3.3 | Parameters of the block reading of participation (rule closed 3 October 2026, section 3.3 Q2 and 3.4.1; ledger F3): the per-block vote bound and the carriage window, and the simulation ran with the narrower cert reading | Add vote carriage in blocks and a hostile aggregator to `finality_v2.py`; re-run A, C, D and F1 under the block reading; set the per-block vote bound and confirm the carriage window of 240 indices | 3 |
| O-3.4 | Certificate grace value; must be at least 3x the worst honest one-way delay (section 3.3, Q4) | Measure one-way delays on the devnet across regions; set grace | 3 |
| O-3.5 | VRF construction for aggregator selection and the binomial sub-user sortition above 8,192 voters (S1, S2) | Specify (candidate: BLS-based VRF on the vote key, Algorand's binomial sampling); simulate the threshold on sampled weight | 3 |
| O-3.6 | What a node does with two valid certificates at one index after a partition heals; post-heal fork choice is unmodelled (`sim/results_v2.md`, "cannot tell us") | Adopt or replace the proposal in section 3.5; devnet partition-and-heal test | 3 |
| O-3.6 | What a node does with two valid certificates at one index after a partition heals; post-heal fork choice is unmodelled (`sim/results_v2.md`, "cannot tell us"). Narrowed 5 October 2026 night (ledger C4): post-heal fork choice for ONE certified chain is now the certificate-driven reorg of 3.5, implemented and measured on the fast-time harness (`tools/finality-attacks/c4.mjs`, bench-log "the C4 fix"); what is left is the two-certificate case, O-3.17 | Adopt or replace the proposal in section 3.5; devnet partition-and-heal test | 3 |
| O-3.7 | The eclipse case is closed by the quorum floor (section 3.3.2, 3 October 2026; ledger F2): 0 conflicting locks at 1, 2 and 4 h against a 34% attacker in the model, but the model grants the attacker the eclipse for free | Devnet with a single-node eclipse recording whether conflicting locks appear, as confirmation of the rule; no rule choice remains | 3 |
| O-3.8 | The simulation has no DAG: conflict counts are index collisions; red blocks, merge under the 3,600-s bound and the finality overlay's effect on GHOSTDAG's guarantees are unmodelled (ledger C4, F8) | Devnet runs with the finality module on and off; a churn and adversary simulation driven by real pool-hashrate traces from mid-cap GPU coins (design document, "Three experiments") | 3 |
| O-3.9 | Model assumptions that move the numbers: perfect or instant DAA retarget (real lag of the order of an hour, approximate), uptime 97% / 99.5% is a guess, silent sets random by key not by pool or region, keys are free, VRF noise absent (`sim/results.md` and `results_v2.md`) | Re-run `finality_v2.py` with a DAA lag model, a top-pool silent set and a regional silent set; price keys through the P2P layer | 3 |

View file

@ -78,6 +78,13 @@ Nothing in consensus changes for any of this: the segment claim already commits
| Records per block | 8 | Implemented, 7.7 (Designed value) |
| Payout per shard | the segment's pool credit in equal parts, remainder to shard 0, paid by the carrying segment | Implemented, 7.7 |
| Proving v0 activation | `proving_v0_activation_daa`, default never | Implemented, 7.7 |
| Segment record (v1) | per segment of `proving_v1_segment_blocks` chain blocks, 586 bytes (the aggregator guest's 340-byte statement inline), BLS-signed by the aggregator's vote key, in the coinbase extra data before the shard record section (`IGNS`); proof bytes on p2p message 75 (protocol 15) | Implemented, 7.8 (branch `proving-v1`, 5 October 2026) |
| Segment records per block | 2 | Implemented, 7.8 (Designed value) |
| Segment length `N` | `proving_v1_segment_blocks`, 8 | Implemented, value Decided (5 October 2026, delegated; `docs/plans/proving-v1.md`) |
| Unproven deadline `T` | `proving_v1_unproven_daa`, 600 DAA s after the segment's last chain block | Implemented, value Decided (5 October 2026, delegated) |
| Aggregator share | `proving_v1_aggregator_share_bps`, 1,000 (a tenth of every attested block's pool credit; the shards share the rest) | Implemented, value Decided (5 October 2026, delegated) |
| Proving v1 activation | `proving_v1_activation_daa`, default never; the segment grid starts at the first chain block at or above it | Implemented, 7.8 |
| Mandatory proofs | the rule of 7.8 item 9, no switch yet, off | Designed |
## 7.5 Proving gas per transaction: the cap and the abort
@ -112,3 +119,20 @@ Added 4 October 2026 because the implementation (`vendor/igneum-node-proving`, b
8. **The plan.** Every segment has at least one shard; an empty segment is one shard whose statement applies the rewards and payouts only. The node cuts from its own per-transaction boundaries (`TxBoundary`: the carry of 7.6, the state root after every transaction) with the cut of `igneum_prove_core::plan`; `igneum-prove-export` must reproduce the node's plan on the same export, and the test network checks that it does.
RPCs (the execution layer's JSON-RPC): `igneum_getShardPlan(block)`, `igneum_getProofRecords(block)`, `igneum_submitProofRecord({record, proof})`, `igneum_getAssignedShards([keyHash...], lookback)` (the prover's work list), `igneum_getProvingStatus()`. Signing without the BLS key material in the prover process: `igneum-miner sign-record` and `key-hash`.
## 7.8 Segment records, the chain rule and the unproven rule, as implemented (proving v1)
Added 5 October 2026 (the project lead: "open the proving round asap"; branch `proving-v1` of the fork and of the main repository, `docs/plans/proving-v1.md`). Every item is Implemented on the branch and behind `proving_v1_activation_daa` (default never); nothing here changes the devnet until the 0.3.11 rollout sets the switch. Item 9 is Designed and off. Proving v0 (7.7) keeps running underneath: per-shard records stay valid and paid.
1. **The aggregated proof.** The aggregator guest of 7.6 (pinned, `elf/igneum-prove-aggregator`) verifies every shard proof of one chain block and, by recursion, the previous chain block's aggregated proof (`AggInput.prev`): its public values (`BlockOutput`, 340 bytes, mirrored in consensus as `BlockStatement`) carry `chain_len`, the number of consecutive chain blocks the proof attests, and `agg_vk`, the aggregator's own id whenever `chain_len > 1`. One compressed proof of constant size therefore attests any run of consecutive chain blocks (measured: `docs/bench-log.md`, "proving v1, aggregated chains on the RTX 5090"). This is design 5.3's "segment N verifies N-1" realised inside the proof rather than beside it.
2. **Segments.** From the first chain block `A` whose DAA score reaches `proving_v1_activation_daa`, chain blocks are grouped in fixed segments of `N = proving_v1_segment_blocks`: segment `k` is `A + kN ..= A + kN + N - 1`. The grid is a pure function of the chain, so every node names the same segments.
3. **The record.** One `SegmentRecord` (`kaspa_consensus_core::proving`): version 2, `first`, `last`, the hash of chain block `last`, the aggregator's BLS vote key, the payout address, the aggregated proof's 340 public values inline, the SHA-256 of the proof bytes, and a BLS signature over all of it under `IGNEUM_SEGMENT_RECORD_V1`, domain-separated with the network name. 586 bytes. The statement the verifier checks is keccak256 of the public values.
4. **Carriage.** Records ride in the coinbase extra data as a section `records || len_le32 || "IGNS"` placed before the shard record section (`IGNP`), which sits before the finality section; at most 2 per block. A node that does not read the section sees miner bytes. The proof bytes travel beside the record on p2p message `IgneumSegmentRecordMessage` (type 75, protocol version 15; peers at 14 and below never receive it; 14 is the EVM transaction relay of 0.3.10) and through `igneum_submitSegmentRecord`.
5. **What every node checks on a carried record (consensus).** The segment is aligned on the grid and its last block is on the executor's own chain, at most 600 chain blocks behind the carrier; the signature verifies; the public values equal the node's native block statement for chain block `last` in every field but `provers` and `chain_len` (`chain_id`, `number`, `block_hash`, `parent_hash`, `shard_count`, `tx_commitment`, `pre_root`, `post_root`, the keccak of the shard receipts roots, gas, pgas, executed, skipped, `shard_vk` = the pinned shard program id, `agg_vk` = the pinned aggregator id when `chain_len > 1` and zero otherwise); `chain_len` is at least `N` and at most the chain height; the chain rule of item 6; and the deadline of item 7. A record that fails is ignored, not a fault of the block: it pays nothing. The SP1 proof is not verified on this path (item 8).
6. **The chain rule.** A record whose `chain_len > N` chains to the previous segment's proof by recursion (the proof itself verified it), and is valid whatever the chain says about that segment. A record whose `chain_len = N` starts a fresh chain and is valid only for the first segment of the grid, or when the previous segment is unproven (item 7). So a proven segment is followed only by records that verify it; an unproven one may be skipped.
7. **The unproven rule.** A segment whose last chain block is more than `T = proving_v1_unproven_daa` DAA seconds old at the carrier, with no paid record, is unproven: the pool pays nothing for it (its aggregator share stays in the escrow), a record for it carried after the deadline is invalid, and the next segment may start a fresh chain. A segment with a paid record is proven; before its deadline it is pending. `igneum_getProvingStatus` reports the three counts over the record window.
8. **Verification is off the consensus path**, as 7.7 item 4: the proof pool verifies segment proofs through `igneum-prove-host --mode verify-segment` (SP1's light verifier against the pinned aggregator key; the shard id and the aggregator id inside the statement checked against the pinned ids; keccak of the public values against the statement) and a producer offers only verified records to its templates. The native statement bounds what an unverified record can do to the aggregator's payout, never to state.
9. **Payout.** From the activation, a chain block's pool credit (5.3) splits: `proving_v1_aggregator_share_bps` of it to the aggregator of the segment record that attests the block, the rest to its shards in equal parts as 7.7 item 6 (`split_pool_credit`). At the carrying chain block, for every segment record its blocks carry in sequence order, the first valid record per segment pays the sum of the aggregator shares of the segment's blocks to the record's payout address, from the escrow, after the shard payouts and before the transactions; a reorg unwinds it with the carrier. There is no aggregator sortition yet: the first valid record carried wins (Open, O-7.3: the VRF draw of design 5.3).
10. **Mandatory proofs (Designed, off).** The rule that makes a proof a condition of validity: a chain block is invalid if the segment ending `T` DAA seconds before it is unproven. It needs an activation height of its own (not yet a parameter) and a consensus-level check in the block validator; it is written here so the devnet measures the coverage the fleet can meet first (`docs/plans/proving-v1.md`, step 3) and the project lead sets the height when the share is one.
RPCs: `igneum_submitSegmentRecord({record, proof})`, `igneum_getSegmentStatement(block)` (the segment, its status, the native public values for a fresh and a continuing chain, the shard proofs the pool holds per block, the previous paid record), `igneum_getSegmentRecords(block)`, `igneum_getProofBytes(block, shard, keyHash)` and `igneum_getSegmentProofBytes(first, keyHash)` (the aggregator's inputs), the `v1` object of `igneum_getProvingStatus`. Signing: `igneum-miner sign-segment-record`. Host modes: `chain`, `aggregate`, `verify-segment`.

View file

@ -20,7 +20,8 @@ use std::sync::Mutex;
use std::time::Instant;
use igneum_pow::accept::{check as accept_check, Reject};
use igneum_pow::generator::{candidate_from_words, generate_v1_from_words, GeneratorConfig, Instr, Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LANES, OP_WEIGHTS};
use igneum_pow::generator::{candidate_from_words, generate_v1_from_words, GeneratorConfig, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LANES, OP_WEIGHTS};
use igneum_pow::memhard::Layout;
use igneum_pow::seed::{fnv1a64, program_rng, seed_words, SplitMix64};
use igneum_pow::verify::{dataset_elem, hash_warp, splitmix32, DatasetMode, DatasetSource};
@ -180,9 +181,9 @@ fn generate_fixed16(seed_string: &str, seed: [u32; 8], fresh: Fresh) -> Program
fresh_reg[src as usize] = false;
}
fresh_reg[dst as usize] = true;
instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask });
instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width: 1, win: 0, off: 0 });
}
Program { seed_string: seed_string.to_string(), seed_bytes: seed_string.as_bytes().to_vec(), seed, generator: GENERATOR_VERSION, attempt: 0, instrs }
Program { seed_string: seed_string.to_string(), seed_bytes: seed_string.as_bytes().to_vec(), seed, generator: GENERATOR_VERSION, attempt: 0, class: LoadClass::V2, era_bytes: None, instrs }
}
// ---------------------------------------------------------------------------------------------------------
@ -478,6 +479,9 @@ fn run_warp(p: &Program, base: u32, ds: &DatasetSource, acc: &mut Acc, lane_addr
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
// Experiment-only ops of the read-width and scratch variants; the census generates the lottery
// hash's class only, so neither op ever appears here.
Op::Scratch | Op::Hot => unreachable!("the census never generates a scratch or hot op"),
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
@ -505,7 +509,7 @@ fn run_warp(p: &Program, base: u32, ds: &DatasetSource, acc: &mut Acc, lane_addr
}
match memhard {
Some(m) => {
m.fetch(&idx, &mut val);
m.fetch(&idx, &mut val, Layout::LINEAR);
}
None => {
for lane in 0..LANES {

View file

@ -11,12 +11,14 @@
//! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) |
//!
//! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and
//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the
//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by
//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the
//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES};
use crate::seed::{fnv1a64, SplitMix64};
use crate::verify::{dataset_elem, splitmix32};
use crate::memhard::hot_index;
use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel};
/// Units (32-lane warps) the dynamic test interprets.
pub const ACCEPT_UNITS: usize = 64;
@ -33,6 +35,14 @@ pub const BIAS_TOLERANCE: u32 = 136;
/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
pub const MIN_DISTINCT_SUM: u64 = 245_760;
/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
pub fn min_distinct_sum(loads: usize) -> u64 {
loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
}
/// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the
/// first failing test in the order of the module table.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
@ -67,7 +77,7 @@ impl std::fmt::Display for Reject {
Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
Reject::DistinctAddresses { sum } => {
write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120)", *sum as f64 / 2048.0)
write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
}
}
}
@ -170,6 +180,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
let seed = &p.seed;
let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1;
let (d0, d1) = (seed[0], seed[1]);
let (h0, h1) = (seed[2], seed[3]);
let hot_words = p.hot_words();
let loads = p.loads_per_hash();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
@ -183,12 +195,30 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
let mut idx = [0u32; LANES];
let mut nload = 0usize;
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
let slot_mask = p.class.scratch_slot_mask();
let era = p.class.era;
for it in 0..ITERATIONS {
let sel = r[0];
for (k, ins) in p.instrs.iter().enumerate() {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Scratch => {
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
for lane in 0..LANES {
idx[lane] = r[a][lane] & slot_mask;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]);
lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane];
}
nload += 1;
}
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
@ -255,18 +285,45 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
}
Op::Load => {
// Read-width experiment: a load of `width` words reads from the aligned address and folds every
// word (verify::fold_words); width 1 is the lottery hash's xor of one word.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = r[a][lane] & mask;
idx[lane] = load_index(era.as_ref(), ins, r[a][lane], mask, ACCEPT_DATASET_LOG2) & align;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
if width == 1 {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
} else {
let mut w = [0u32; 16];
for j in 0..width {
w[j] = dataset_elem(idx[lane] + j as u32, d0, d1);
}
r[d][lane] = fold_words(r[d][lane], &w[..width]);
}
lane_addrs[lane * loads + nload] = idx[lane];
}
nload += 1;
}
Op::Hot => {
// Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with
// bit 30 so a hot word and a dataset word at one index count as two addresses.
for lane in 0..LANES {
idx[lane] = hot_index(r[a][lane], hot_words);
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], h0, h1);
lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane];
}
nload += 1;
}
Op::WLoad => {
let b = (r[a][0] & mask) & !31;
for lane in 0..LANES {
@ -298,7 +355,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
sl.sort_unstable();
let mut distinct = 0u64;
for k in 0..loads {
if k == 0 || sl[k] != sl[k - 1] {
// scratch slots carry bit 31 (variant 5) and are not dataset addresses
if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
distinct += 1;
}
}
@ -333,7 +391,7 @@ pub fn check_dynamic(p: &Program) -> Result<AcceptReport, Reject> {
}
bias_max = bias_max.max(d);
}
if acc.distinct_sum <= MIN_DISTINCT_SUM {
if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
}
Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
@ -348,9 +406,50 @@ pub fn check(p: &Program) -> Result<AcceptReport, Reject> {
#[cfg(test)]
mod tests {
use super::*;
use crate::generator::{candidate, generate, GeneratorConfig, generate_v1};
use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass};
use crate::verify::{DatasetMode, DatasetSource};
#[test]
fn distinct_bound_scales_with_the_load_count() {
assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM);
assert_eq!(min_distinct_sum(32), 61_440);
}
/// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees
/// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
#[test]
fn classes_pass_and_match_verify() {
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-rw-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
let bases = accept_base_nonces(&p.seed);
let loads = p.loads_per_hash();
let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut ones = [0u32; 64];
for (u, &b) in bases.iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for h in crate::verify::hash_warp(&p, b, &ds) {
for j in 0..64 {
ones[j] += ((h >> j) & 1) as u32;
}
}
}
assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter");
}
}
/// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words.
#[test]
fn instrumented_interpreter_matches_verify() {
@ -381,6 +480,52 @@ mod tests {
}
}
/// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are
/// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads).
#[test]
fn hot_classes_pass_and_hot_loads_are_uniform() {
for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-hot-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
}
let p = generate_class("igneum-genesis", LoadClass::hot(96, 4));
let words = p.hot_words();
let loads = p.loads_per_hash();
let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut buckets = [0u64; 16];
let mut hot_count = 0u64;
for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for &a in &la {
if a & 0xC000_0000 == 0x4000_0000 {
let idx = a & 0x3FFF_FFFF;
assert!(idx < words);
buckets[(idx as u64 * 16 / words as u64) as usize] += 1;
hot_count += 1;
}
}
}
assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes");
let mean = hot_count as f64 / 16.0;
for (i, &b) in buckets.iter().enumerate() {
assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}");
}
// the dataset distinct count still holds for the dataset loads alone
let r = check(&p).unwrap();
assert!(r.distinct_mean() > 120.0);
}
#[test]
fn base_nonces_are_aligned_and_seed_dependent() {
let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]);

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

Some files were not shown because too many files have changed in this diff Show more