From a6013214816440f61d83b573ebdab308374e165c Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Sun, 4 Oct 2026 19:20:49 +0000 Subject: [PATCH 1/3] App engine: watchdog per card and per node, WORKER FAULT and mismatched= read from the miner; public GPU bench table and the reliability test harness Review round 4 X21 and the project lead's levers 5 and 6 (4 October 2026 evening): - src/watchdog.rs: CardWatch (no status line for 90 s, or hash rate 0 for 60 s while the node is synced: one restart, then the card is marked faulted with the reason in the UI and the log; the miner's own worker restarts suspend both rules until ready, so there are no double restarts; exit 43 counts as a watchdog restart) and NodeWatch (our node silent for 120 s is restarted in-process through the restart kind the remote jobs use, with a growing delay). State machines with no clock; 11 unit tests replay recorded STATUS and WORKER FAULT lines. - engine.rs reads STATUS mismatched=, faults= and the WORKER FAULT lines onto the card (faults, mismatched, message); a faulted card is not restarted until the user changes its settings or resumes mining; other cards keep mining. - site/miners.html (build.mjs, scrubbed like /bench, rows from site/miner-bench.json): card, generator version, best MH/s, MH per watt where measured, miner version, date, source, measured by the team or reported by the fleet. In the navigation beside the engineering log on every page and in the sitemap. One line says there is no other miner to compare with. - tools/reliability: fake-worker.mjs (the serve protocol, misbehaving on command), run.mjs (miner guards on a private test network, ports 29950+) and app-run.mjs (the app watchdog end to end, ports 29960+). Co-Authored-By: Claude Fable 5.1 --- app/igneum-app/src/engine.rs | 176 +++++++++-- app/igneum-app/src/main.rs | 1 + app/igneum-app/src/state.rs | 6 +- app/igneum-app/src/watchdog.rs | 482 ++++++++++++++++++++++++++++++ app/igneum-app/ui/app.js | 4 +- site/bench.html | 2 +- site/build.mjs | 29 +- site/evidence.html | 2 +- site/index.html | 2 + site/litepaper.html | 2 +- site/live.html | 2 + site/miner-bench.json | 71 +++++ site/miners.html | 85 ++++++ site/sitemap.xml | 1 + tools/reliability/app-run.mjs | 217 ++++++++++++++ tools/reliability/fake-worker.mjs | 74 +++++ tools/reliability/run.mjs | 297 ++++++++++++++++++ 17 files changed, 1423 insertions(+), 30 deletions(-) create mode 100644 app/igneum-app/src/watchdog.rs create mode 100644 site/miner-bench.json create mode 100644 site/miners.html create mode 100755 tools/reliability/app-run.mjs create mode 100755 tools/reliability/fake-worker.mjs create mode 100755 tools/reliability/run.mjs diff --git a/app/igneum-app/src/engine.rs b/app/igneum-app/src/engine.rs index fe4b8e20..81fdd6ef 100644 --- a/app/igneum-app/src/engine.rs +++ b/app/igneum-app/src/engine.rs @@ -356,6 +356,8 @@ struct MinerSlot { prepared: bool, // the pack is exported (and the worker built) for the next start last_status: Option, error_at: Option, + /// the per-card watchdog (src/watchdog.rs): one restart, then faulted + watch: crate::watchdog::CardWatch, } pub struct Engine { @@ -416,6 +418,10 @@ pub struct Engine { last_stability: Instant, last_settings_save: Instant, last_error_event: Instant, + /// the node watchdog (src/watchdog.rs): a silent node of ours is restarted + node_watch: crate::watchdog::NodeWatch, + /// the watchdogs' clock origin (they take seconds) + t0: Instant, } impl Engine { @@ -487,6 +493,8 @@ impl Engine { last_stability: now, last_settings_save: now, last_error_event: now - Duration::from_secs(600), + node_watch: crate::watchdog::NodeWatch::new(), + t0: now, } } @@ -494,6 +502,11 @@ impl Engine { self.shared.state.lock().unwrap() } + /// Seconds on the watchdogs' clock. + fn secs(&self, now: Instant) -> f64 { + now.duration_since(self.t0).as_secs_f64() + } + pub fn run(mut self) { let setup_done = self.shared.settings.lock().unwrap().setup_done; self.shared.log(&format!( @@ -608,6 +621,13 @@ impl Engine { self.shared.save_settings(); self.st().mining.paused = false; self.shared.event("ok", "mining resumed"); + // a user action: faulted cards try again + for m in self.miners.iter_mut() { + if m.watch.faulted().is_some() { + m.watch.event(0.0, crate::watchdog::Event::Reset); + m.restart_at = Some(Instant::now()); + } + } if !self.running { self.start(); } @@ -616,6 +636,8 @@ impl Engine { self.shared.event("info", &format!("settings changed ({why}); the miners restart")); self.stop_miners("settings changed"); for m in self.miners.iter_mut() { + // a user action: a faulted card tries again + m.watch.event(0.0, crate::watchdog::Event::Reset); m.restart_at = Some(Instant::now()); } } @@ -958,6 +980,7 @@ impl Engine { prepared: false, last_status: None, error_at: None, + watch: crate::watchdog::CardWatch::new(), }); } if self.miners.is_empty() { @@ -1059,6 +1082,8 @@ impl Engine { self.miners[i].proc = Some(p); self.miners[i].restart_at = None; self.miners[i].last_status = None; + let t = self.secs(Instant::now()); + self.miners[i].watch.event(t, crate::watchdog::Event::Started); } Err(e) => { self.shared.event("error", &format!("the miner could not start: {e}")); @@ -1072,17 +1097,19 @@ impl Engine { if any { self.shared.log(&format!("stopping the miners ({why})")); } + let t = self.secs(Instant::now()); for m in self.miners.iter_mut() { if let Some(mut p) = m.proc.take() { p.write_stdin("quit\n"); p.stop(8); } m.restart_at = None; + m.watch.event(t, crate::watchdog::Event::Stopped); } let mut st = self.st(); let paused = st.mining.paused; for c in st.mining.cards.iter_mut() { - if c.enabled { + if c.enabled && c.state != "faulted" { c.state = if paused { "off".into() } else { "waiting".into() }; c.hash_now = 0.0; c.pid = 0; @@ -1154,6 +1181,7 @@ impl Engine { if enabled { self.miners[pos].restart_at = Some(Instant::now()); self.miners[pos].prepared = false; + self.miners[pos].watch.event(0.0, crate::watchdog::Event::Reset); self.shared.event("info", &format!("{name}: {identities} identit{}; its worker restarts", if identities == 1 { "y" } else { "ies" })); } else { self.miners.remove(pos); @@ -1477,13 +1505,7 @@ impl Engine { if self.node_external { self.shared.event("info", "remote job asked for a node restart, but the node is external; nothing done"); } else { - self.stop_miners("remote job: restart node"); - self.stop_node(); - self.node_restart_at = Some(Instant::now() + Duration::from_secs(3)); - for m in self.miners.iter_mut() { - m.restart_at = Some(Instant::now()); - } - self.st().node.message = "restart asked by a remote job".into(); + self.restart_node("restart asked by a remote job", Duration::from_secs(3)); } } Action::RestartApp => { @@ -1635,6 +1657,15 @@ impl Engine { st.node.message = "no answer from the RPC yet (still opening its database?)".into(); } } + // the node watchdog: our node silent for 120 s (no reading, nothing accepted) is restarted in-process + if self.node.is_some() && !self.node_external && self.node_restart_at.is_none() { + let silent_s = self.node_last_reading.map(|t| now.duration_since(t)).unwrap_or_else(|| self.node_started_at.elapsed()).as_secs_f64(); + if let Some(delay) = self.node_watch.tick(self.secs(now), true, silent_s, accepted_recent) { + let n = self.node_watch.restarts_in_window(); + self.shared.event("error", &format!("no reading from the node for {} s; the watchdog restarts it in {delay} s (restart {n} in the last 10 minutes)", silent_s as u64)); + self.restart_node("no reading from the node for 120 s", Duration::from_secs(delay)); + } + } } fn tick_miners(&mut self, now: Instant) { @@ -1648,12 +1679,17 @@ impl Engine { let code = p.exit_code.unwrap_or(-1); let tail = tail_of(&p.log_path, 3); self.miners[i].proc = None; + let t = self.secs(now); + let verdict = self.miners[i].watch.event(t, crate::watchdog::Event::Exited(code)); if code == 42 { self.shared.event("build", "the hourly program changed; the worker restarts on the new one"); if self.bins_worker_missing(card_idx) { self.miners[i].needs_rebuild = true; } self.miners[i].restart_at = Some(now); + } else if verdict != crate::watchdog::Action::None { + // exit 43: the miner gave up on its worker; once more, then the card is faulted + self.watchdog_verdict(i, verdict, &tail); } else { self.miners[i].restarts += 1; let delay = jitter_secs(crate::platform::unix_now() + i as u64); @@ -1669,17 +1705,34 @@ impl Engine { c.message = tail.last().cloned().unwrap_or_default(); } } - } else if let Some(t) = self.miners[i].last_status { - if now.duration_since(t) > Duration::from_secs(self.shared.runtime.status_secs as u64 * 4) { - if let Some(c) = self.st().mining.cards.get_mut(card_idx) { - if c.state == "mining" { - c.message = "no status from the miner for a while".into(); + } else { + // the watchdog: no status for 90 s, or a zero rate for 60 s while synced, is one restart, then faulted + let t = self.secs(now); + let verdict = self.miners[i].watch.tick(t, synced); + if verdict != crate::watchdog::Action::None { + if let Some(mut p) = self.miners[i].proc.take() { + p.write_stdin("quit\n"); + p.stop(3); + } + self.watchdog_verdict(i, verdict, &[]); + continue; + } + if let Some(t) = self.miners[i].last_status { + if now.duration_since(t) > Duration::from_secs(self.shared.runtime.status_secs as u64 * 4) { + if let Some(c) = self.st().mining.cards.get_mut(card_idx) { + if c.state == "mining" { + c.message = "no status from the miner for a while".into(); + } } } } } continue; } + if self.miners[i].watch.faulted().is_some() { + // faulted by the watchdog: not restarted until the user changes the card's settings or resumes + continue; + } if self.job_hold { // a remote job has the GPU; the miners wait until it lets go (src/jobrun.rs) continue; @@ -1711,6 +1764,58 @@ impl Engine { } } + /// Applies a watchdog verdict to slot `i` (its process already stopped or gone): a restart now with the reason on + /// the card, or the card marked faulted with the reason in the UI and the log while the other cards keep mining. + fn watchdog_verdict(&mut self, i: usize, verdict: crate::watchdog::Action, tail: &[String]) { + use crate::watchdog::Action; + let card_idx = self.miners[i].card; + let name = self.st().mining.cards.get(card_idx).map(|c| c.name.clone()).unwrap_or_else(|| self.miners[i].label.clone()); + for l in tail { + self.shared.log(&format!(" miner: {l}")); + } + match verdict { + Action::Restart(reason) => { + self.miners[i].restarts += 1; + self.miners[i].restart_at = Some(Instant::now() + Duration::from_secs(1)); + self.shared.event("error", &format!("{name}: {reason}; the watchdog restarts the miner (restart {} of 1 before the card is marked faulted)", self.miners[i].watch.watchdog_restarts())); + if let Some(c) = self.st().mining.cards.get_mut(card_idx) { + c.state = "restarting".into(); + c.restarts = self.miners[i].restarts; + c.hash_now = 0.0; + c.pid = 0; + c.message = format!("watchdog: {reason}"); + } + } + Action::Fault(reason) => { + self.miners[i].restart_at = None; + self.shared.event("error", &format!("{name}: {reason}; the card is marked faulted and its miner is not restarted again (the other cards keep mining; change the card's settings or resume mining to try again)")); + if let Some(c) = self.st().mining.cards.get_mut(card_idx) { + c.state = "faulted".into(); + c.hash_now = 0.0; + c.pid = 0; + c.restart_in_s = 0; + c.message = format!("faulted: {reason}"); + } + } + Action::None => {} + } + } + + /// Stops the miners and the node and schedules the node's start after `delay`; the miners follow once it is + /// synced. The remote-job restart kind and the node watchdog share it. + fn restart_node(&mut self, why: &str, delay: Duration) { + self.stop_miners(why); + self.stop_node(); + self.node_restart_at = Some(Instant::now() + delay); + for m in self.miners.iter_mut() { + m.restart_at = Some(Instant::now()); + } + let mut st = self.st(); + st.node.state = "restarting".into(); + st.node.synced = false; + st.node.message = format!("{why}; restart in {} s", delay.as_secs()); + } + fn bins_worker_missing(&self, card_idx: usize) -> bool { let st = self.st(); match st.mining.cards.get(card_idx).map(|c| c.worker.as_str()) { @@ -1944,7 +2049,7 @@ impl Engine { if synced && !was_synced { self.shared.event("ok", &format!("node synced: {blocks} blocks, {peers} peer(s)")); for m in self.miners.iter_mut() { - if m.proc.is_none() && m.restart_at.is_none() { + if m.proc.is_none() && m.restart_at.is_none() && m.watch.faulted().is_none() { m.restart_at = Some(now); } } @@ -1970,19 +2075,38 @@ impl Engine { if let Some(c) = self.st().mining.cards.get_mut(card) { c.message = "the node is not answering; the miner retries".into(); } - } else if text.contains("WORKER MISMATCH") || text.contains("worker error") || text.contains("worker exited") || text.contains("panicked") || text.contains("CUDA error") || text.contains("submit error") { + } else if text.contains("WORKER MISMATCH") || text.contains("worker error") || text.contains("worker exited") || text.contains("worker killed by a guard") || text.contains("panicked") || text.contains("CUDA error") || text.contains("submit error") { let now = Instant::now(); if now.duration_since(self.last_error_event) >= Duration::from_secs(30) { self.last_error_event = now; self.shared.event("error", &format!("{card_name}: {}", short(text, 160))); } self.miners[i].error_at = Some(now); + if text.contains("worker exited") || text.contains("worker killed by a guard") { + // the miner restarts its worker itself: the watchdog waits for `ready` instead of restarting the miner too + let t = self.secs(now); + let reason = short(text.split_once(' ').map(|(_, r)| r).unwrap_or(text), 160); + self.miners[i].watch.event(t, crate::watchdog::Event::WorkerRestart(&reason)); + } if let Some(c) = self.st().mining.cards.get_mut(card) { c.message = short(text, 160); } } return; } + if text.contains(" WORKER FAULT ") { + // the miner's guards (interval, job time, cpu re-check, stall) killed the worker; it restarts it itself + let reason = crate::watchdog::fault_reason(text).unwrap_or_else(|| short(text, 160)); + let t = self.secs(Instant::now()); + self.miners[i].watch.event(t, crate::watchdog::Event::WorkerRestart(&reason)); + self.shared.event("error", &format!("{card_name}: worker fault: {}; the miner restarts the worker", short(&reason, 200))); + if let Some(c) = self.st().mining.cards.get_mut(card) { + c.faults += 1; + c.hash_now = 0.0; + c.message = format!("worker fault: {}", short(&reason, 160)); + } + return; + } if text.contains(" ACCEPTED block") { self.last_accepted = Some(Instant::now()); { @@ -2009,27 +2133,33 @@ impl Engine { } drop(st); self.shared.event("block", &format!("block accepted by the node ({card_name})")); - } else if text.contains(" STATUS '") { - self.miners[i].last_status = Some(Instant::now()); + } else if let Some(s) = crate::watchdog::parse_status(text) { + let now = Instant::now(); + self.miners[i].last_status = Some(now); + let t = self.secs(now); + self.miners[i].watch.event(t, crate::watchdog::Event::Status(&s)); let avg = kv_f64(text, "hash").unwrap_or(0.0); - // the v4 miner's interval rate: "now=12.34 MH/s wall (...)"; older miners: the average - let now_rate = kv_f64(text, "now").unwrap_or(avg); let mut st = self.st(); if let Some(c) = st.mining.cards.get_mut(card) { c.hash_avg = avg; - c.hash_now = now_rate; + // the v4 miner's interval rate: "now=12.34 MH/s wall (...)"; older miners: the average + c.hash_now = s.hash_now; c.template_age_s = kv_f64(text, "template_age").unwrap_or(0.0); - c.synced = kv(text, "synced") == Some("true"); + c.synced = s.synced; + c.mismatched = s.mismatched; + c.faults = c.faults.max(s.faults); c.last_status_age_s = 0.0; if c.state == "starting" || c.state == "ready" { c.state = "mining".into(); } - if c.state == "mining" { - c.message = String::new(); + if c.state == "mining" && s.hash_now > 0.0 { + c.message = if s.mismatched > 0 { format!("{} share(s) failed the CPU re-check this run", s.mismatched) } else { String::new() }; } } } else if text.contains(" worker: ready ") { let prepare = text.contains(" prepare 1"); + let t = self.secs(Instant::now()); + self.miners[i].watch.event(t, crate::watchdog::Event::Ready); let mut st = self.st(); if let Some(c) = st.mining.cards.get_mut(card) { c.ready = true; diff --git a/app/igneum-app/src/main.rs b/app/igneum-app/src/main.rs index 48c74998..4db19535 100644 --- a/app/igneum-app/src/main.rs +++ b/app/igneum-app/src/main.rs @@ -25,6 +25,7 @@ mod update; mod jobs; mod jobrun; mod prover; +mod watchdog; use std::io::{BufRead, Write}; use std::sync::mpsc::channel; diff --git a/app/igneum-app/src/state.rs b/app/igneum-app/src/state.rs index b5b25b55..1bc64db1 100644 --- a/app/igneum-app/src/state.rs +++ b/app/igneum-app/src/state.rs @@ -40,7 +40,7 @@ pub struct CardState { pub detail: String, // memory, cores pub device: String, // the worker's --device value (Windows) pub enabled: bool, - pub state: String, // off | waiting | starting | ready | mining | restarting | failed + pub state: String, // off | waiting | starting | ready | mining | restarting | failed | faulted (the watchdog gave up on it) pub hash_now: f64, // MH/s, the last interval pub hash_avg: f64, // MH/s since the start pub accepted: u64, @@ -57,6 +57,10 @@ pub struct CardState { pub pid: u32, pub last_status_age_s: f64, pub message: String, + /// `WORKER FAULT` lines seen (the miner killed and restarted its worker) and the miner's own `faults=` count + pub faults: u64, + /// shares that failed the miner's CPU re-check this run (`mismatched=` on the STATUS line) + pub mismatched: u64, // power and heat (NVIDIA through nvidia-smi; 0 = unknown) pub power_w: f64, // draw now pub power_limit_w: f64, // the limit in force diff --git a/app/igneum-app/src/watchdog.rs b/app/igneum-app/src/watchdog.rs new file mode 100644 index 00000000..27f30c83 --- /dev/null +++ b/app/igneum-app/src/watchdog.rs @@ -0,0 +1,482 @@ +//! The engine's watchdog, as state machines with no clock and no process (seconds come from the caller), so the +//! rules can be unit-tested with recorded miner lines. +//! +//! Per card (`CardWatch`): a miner that prints no status line for 90 s, or reports a hash rate of 0 for 60 s while +//! the node is synced, is restarted once; when that recurs before five minutes of healthy status, the card is marked +//! faulted with the reason, its miner is not restarted again, and the other cards keep mining. The miner's own +//! worker restarts (`WORKER FAULT`, `worker exited`) suspend both rules until the worker is ready again, so the app +//! never restarts a miner that is already restarting its worker (no double restarts); if the worker is not back +//! within 180 s the app steps in. Exit code 43 (the miner gave up on its worker after three guard trips) counts +//! like a watchdog restart: once, then faulted. +//! +//! Per node (`NodeWatch`): a node of ours that answers no `watch` reading for 120 s is restarted by the app, with a +//! growing delay when it repeats inside ten minutes. +//! +//! Review round 4, X21 (4 October 2026): the app now reads `mismatched=` and `faults=` from STATUS lines and the +//! `WORKER FAULT` lines, and shows them on the card. + +/// No status line from a running miner for this long: restart it. +pub const NO_STATUS_S: f64 = 90.0; +/// Hash rate 0 while the node is synced and the worker is ready for this long: restart the miner. +pub const ZERO_RATE_S: f64 = 60.0; +/// The miner is restarting its own worker: the app waits this long for `ready` before it steps in. +pub const WORKER_RESTART_GRACE_S: f64 = 180.0; +/// Continuous healthy status (rate above 0) that renews the one-restart budget. +pub const HEALTHY_RESET_S: f64 = 300.0; +/// `igneum-miner` exit code when its guards gave up on the worker (three trips in ten minutes). +pub const MINER_GAVE_UP_CODE: i32 = 43; +/// No `watch` reading from our node for this long: restart it. +pub const NODE_SILENT_S: f64 = 120.0; +/// Node restarts inside this window grow the delay before the next one. +pub const NODE_WINDOW_S: f64 = 600.0; + +/// What the watchdog reads from one `STATUS` line of `igneum-miner`. +#[derive(Debug, Clone, PartialEq, Default)] +pub struct Status { + /// MH/s of the last interval (`now=`; older miners: the average `hash=`) + pub hash_now: f64, + pub mismatched: u64, + pub faults: u64, + pub restarts: u64, + pub synced: bool, +} + +/// Parses a miner STATUS line; None for any other line. +pub fn parse_status(text: &str) -> Option { + if !text.contains(" STATUS '") { + return None; + } + let avg = kv_f64(text, "hash").unwrap_or(0.0); + Some(Status { + hash_now: kv_f64(text, "now").unwrap_or(avg), + mismatched: kv_u64(text, "mismatched").unwrap_or(0), + faults: kv_u64(text, "faults").unwrap_or(0), + restarts: kv_u64(text, "restarts").unwrap_or(0), + synced: kv(text, "synced") == Some("true"), + }) +} + +/// The reason text of a `WORKER FAULT` line, without the timestamp and the trailing restart note. +pub fn fault_reason(text: &str) -> Option { + let i = text.find("WORKER FAULT ")?; + let rest = &text["WORKER FAULT ".len() + i..]; + let end = rest.find("; restarting the worker").or_else(|| rest.find("; killing the worker")).unwrap_or(rest.len()); + Some(rest[..end].trim().to_string()) +} + +#[derive(Debug, Clone, PartialEq)] +pub enum Event<'a> { + /// The app started the miner process. + Started, + /// The worker said ready. + Ready, + /// A STATUS line. + Status(&'a Status), + /// The miner killed its worker (a guard) or saw it exit; it restarts the worker itself. + WorkerRestart(&'a str), + /// The miner process ended with this code. + Exited(i32), + /// The app stopped the miner on purpose (pause, settings, node restart). + Stopped, + /// The user changed the card's settings: a faulted card may try again. + Reset, +} + +#[derive(Debug, Clone, PartialEq)] +pub enum Action { + None, + /// Stop the miner and start it again now, for this reason. + Restart(String), + /// Mark the card faulted with this reason; do not restart its miner. + Fault(String), +} + +#[derive(Debug, Default)] +pub struct CardWatch { + started_s: Option, + last_status_s: Option, + ready: bool, + zero_since: Option, + healthy_since: Option, + worker_restart_since: Option, + /// app-level restarts without five healthy minutes since + restarts: u32, + faulted: Option, + last_fault: String, +} + +impl CardWatch { + pub fn new() -> Self { + Self::default() + } + + pub fn faulted(&self) -> Option<&str> { + self.faulted.as_deref() + } + + pub fn watchdog_restarts(&self) -> u32 { + self.restarts + } + + fn arm(&mut self, now_s: f64) { + self.started_s = Some(now_s); + self.last_status_s = None; + self.ready = false; + self.zero_since = None; + self.healthy_since = None; + self.worker_restart_since = None; + } + + fn escalate(&mut self, reason: String) -> Action { + self.started_s = None; + self.last_status_s = None; + self.ready = false; + self.zero_since = None; + self.healthy_since = None; + self.worker_restart_since = None; + if self.restarts >= 1 { + let r = format!("{reason} (restarted once already)"); + self.faulted = Some(r.clone()); + Action::Fault(r) + } else { + self.restarts += 1; + Action::Restart(reason) + } + } + + pub fn event(&mut self, now_s: f64, ev: Event<'_>) -> Action { + match ev { + Event::Started => { + self.arm(now_s); + Action::None + } + Event::Ready => { + self.ready = true; + self.worker_restart_since = None; + Action::None + } + Event::Status(s) => { + self.last_status_s = Some(now_s); + if s.hash_now > 0.0 { + self.zero_since = None; + let since = *self.healthy_since.get_or_insert(now_s); + if now_s - since >= HEALTHY_RESET_S { + self.restarts = 0; + } + } else { + self.healthy_since = None; + if self.ready && self.worker_restart_since.is_none() { + self.zero_since.get_or_insert(now_s); + } + } + Action::None + } + Event::WorkerRestart(reason) => { + self.last_fault = reason.to_string(); + self.worker_restart_since = Some(now_s); + self.ready = false; + self.zero_since = None; + self.healthy_since = None; + Action::None + } + Event::Exited(code) => { + let running = self.started_s.is_some(); + self.started_s = None; + if code == MINER_GAVE_UP_CODE && running { + let why = if self.last_fault.is_empty() { "its guards tripped three times in ten minutes".to_string() } else { self.last_fault.clone() }; + self.escalate(format!("the miner gave up on its worker: {why}")) + } else { + Action::None + } + } + Event::Stopped => { + self.started_s = None; + self.last_status_s = None; + self.ready = false; + self.zero_since = None; + self.worker_restart_since = None; + Action::None + } + Event::Reset => { + *self = Self::default(); + Action::None + } + } + } + + /// Called every engine tick while the miner process is alive. + pub fn tick(&mut self, now_s: f64, node_synced: bool) -> Action { + if self.faulted.is_some() { + return Action::None; + } + let Some(started) = self.started_s else { return Action::None }; + if let Some(t) = self.worker_restart_since { + // the miner is restarting its worker: its own guards own the card until the worker is ready + if now_s - t > WORKER_RESTART_GRACE_S { + return self.escalate(format!("the worker did not come back within {} s of the miner restarting it ({})", WORKER_RESTART_GRACE_S as u64, self.last_fault)); + } + return Action::None; + } + let last = self.last_status_s.unwrap_or(started); + if now_s - last > NO_STATUS_S { + return self.escalate(format!("no status line from the miner for {} s", NO_STATUS_S as u64)); + } + if !node_synced { + // a zero rate while the node syncs is expected; the timer starts again once it is synced + self.zero_since = None; + } + if node_synced && self.ready { + if let Some(z) = self.zero_since { + if now_s - z >= ZERO_RATE_S { + return self.escalate(format!("hash rate 0 for {} s while the node is synced", ZERO_RATE_S as u64)); + } + } + } + Action::None + } +} + +/// The node watchdog: no reading for `NODE_SILENT_S` restarts our node, with a growing delay when it repeats. +#[derive(Debug, Default)] +pub struct NodeWatch { + restarts_s: Vec, +} + +impl NodeWatch { + pub fn new() -> Self { + Self::default() + } + + /// `silent_s`: seconds since the last reading (or since the node started, when it never answered). Returns the + /// delay in seconds before the restart when one is due. + pub fn tick(&mut self, now_s: f64, ours: bool, silent_s: f64, accepted_recent: bool) -> Option { + if !ours || accepted_recent || silent_s < NODE_SILENT_S { + return None; + } + self.restarts_s.retain(|t| now_s - *t <= NODE_WINDOW_S); + self.restarts_s.push(now_s); + let n = self.restarts_s.len(); + Some((3u64.saturating_mul(1u64 << (n.saturating_sub(1) * 2).min(8))).min(300)) + } + + pub fn restarts_in_window(&self) -> usize { + self.restarts_s.len() + } +} + +fn kv<'a>(line: &'a str, key: &str) -> Option<&'a str> { + let pat = format!(" {key}="); + let i = line.find(&pat)? + pat.len(); + let rest = &line[i..]; + let end = rest.find(|c: char| c == ' ' || c == ',' || c == ')' || c == ';').unwrap_or(rest.len()); + Some(&rest[..end]) +} + +fn kv_u64(line: &str, key: &str) -> Option { + kv(line, key)?.parse().ok() +} + +fn kv_f64(line: &str, key: &str) -> Option { + kv(line, key)?.trim_end_matches('s').parse().ok() +} + +#[cfg(test)] +mod tests { + use super::*; + + // Lines as igneum-miner 0.3.x prints them (devnet v4, 4 October 2026; identities and labels shortened). + const STATUS_OK: &str = "1791138616.597 STATUS 'win-1' [worker]: 70s jobs=486 accepted=3 rejected=0 mismatched=0 extra=0 rate=0.04 blocks/s hash=123.90 MH/s wall (124.20 MH/s inside jobs) now=124.10 MH/s wall (124.30 MH/s inside jobs, 70 jobs, seed walk 0 calls) template_age=0.31s synced=true idle=0.2% (last 10s: 0.1%) queued=2 restarts=0 faults=0 identities=8 accepted_by_identity=1/0/1/0/0/1/0/0"; + const STATUS_ZERO: &str = "1791138626.597 STATUS 'win-1' [worker]: 80s jobs=486 accepted=3 rejected=0 mismatched=0 extra=0 rate=0.04 blocks/s hash=108.41 MH/s wall (124.20 MH/s inside jobs) now=0.00 MH/s wall (0.00 MH/s inside jobs, 0 jobs, seed walk 0 calls) template_age=0.31s synced=true idle=12.5% (last 10s: 100.0%) queued=2 restarts=0 faults=0 identities=8 accepted_by_identity=1/0/1/0/0/1/0/0"; + const STATUS_MISMATCH: &str = "1791138636.597 STATUS 'win-1' [worker]: 90s jobs=500 accepted=3 rejected=0 mismatched=3 extra=0 rate=0.03 blocks/s hash=110.00 MH/s wall (124.20 MH/s inside jobs) now=124.00 MH/s wall (124.30 MH/s inside jobs, 70 jobs, seed walk 0 calls) template_age=0.31s synced=true idle=0.2% (last 10s: 0.1%) queued=2 restarts=1 faults=1 identities=8 accepted_by_identity=1/0/1/0/0/1/0/0"; + const FAULT_LINE: &str = "1791138640.100 WORKER FAULT the last 10 s ran at 4300000.00 MH/s inside jobs against 3.30 MH/s over this worker's healthy intervals, over 10x: the worker is not running its kernel; restarting the worker (fault 1)"; + + #[test] + fn parses_status_and_fault_lines() { + let s = parse_status(STATUS_OK).unwrap(); + assert_eq!(s, Status { hash_now: 124.10, mismatched: 0, faults: 0, restarts: 0, synced: true }); + let m = parse_status(STATUS_MISMATCH).unwrap(); + assert_eq!((m.mismatched, m.faults, m.restarts), (3, 1, 1)); + assert_eq!(parse_status(STATUS_ZERO).unwrap().hash_now, 0.0); + assert_eq!(parse_status("1791138640.100 worker: ready metal Apple_M5_Max prepare 1"), None); + assert_eq!(fault_reason(FAULT_LINE).unwrap(), "the last 10 s ran at 4300000.00 MH/s inside jobs against 3.30 MH/s over this worker's healthy intervals, over 10x: the worker is not running its kernel"); + assert_eq!(fault_reason(STATUS_OK), None); + } + + fn healthy(w: &mut CardWatch, from: f64, to: f64) { + let s = parse_status(STATUS_OK).unwrap(); + let mut t = from; + while t <= to { + assert_eq!(w.event(t, Event::Status(&s)), Action::None); + assert_eq!(w.tick(t, true), Action::None); + t += 10.0; + } + } + + #[test] + fn healthy_miner_is_left_alone() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 3600.0); + assert_eq!(w.watchdog_restarts(), 0); + } + + #[test] + fn zero_rate_restarts_once_then_faults() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 100.0); + let z = parse_status(STATUS_ZERO).unwrap(); + for t in [110.0, 120.0, 130.0, 140.0, 150.0, 160.0] { + w.event(t, Event::Status(&z)); + assert_eq!(w.tick(t, true), Action::None, "under 60 s at {t}"); + } + w.event(170.0, Event::Status(&z)); + let a = w.tick(170.0, true); + assert!(matches!(a, Action::Restart(ref r) if r.contains("hash rate 0 for 60 s")), "{a:?}"); + // the app restarted it; still zero: faulted, not restarted again + w.event(172.0, Event::Started); + w.event(174.0, Event::Ready); + for t in [180.0, 190.0, 200.0, 210.0, 220.0, 230.0] { + w.event(t, Event::Status(&z)); + assert_eq!(w.tick(t, true), Action::None); + } + w.event(240.0, Event::Status(&z)); + let a = w.tick(240.0, true); + assert!(matches!(a, Action::Fault(ref r) if r.contains("restarted once already")), "{a:?}"); + assert!(w.faulted().is_some()); + // faulted stays: no more actions + w.event(250.0, Event::Status(&z)); + assert_eq!(w.tick(260.0, true), Action::None); + // the user changed the card's settings: a fresh start + w.event(300.0, Event::Reset); + assert!(w.faulted().is_none()); + } + + #[test] + fn zero_rate_only_counts_with_a_synced_node_and_a_ready_worker() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + let z = parse_status(STATUS_ZERO).unwrap(); + // not ready yet (program loading): no zero timer + for t in [10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 70.0, 80.0] { + w.event(t, Event::Status(&z)); + assert_eq!(w.tick(t, true), Action::None); + } + w.event(82.0, Event::Ready); + // node not synced: no zero timer either + for t in [90.0, 100.0, 110.0, 120.0, 130.0, 140.0, 150.0, 160.0] { + w.event(t, Event::Status(&z)); + assert_eq!(w.tick(t, false), Action::None); + } + assert_eq!(w.tick(170.0, true), Action::None, "the timer starts at the first zero status after ready"); + for t in [180.0, 190.0, 200.0, 210.0, 220.0, 230.0] { + w.event(t, Event::Status(&z)); + w.tick(t, true); + } + assert!(matches!(w.tick(240.0, true), Action::Restart(_))); + } + + #[test] + fn no_status_for_90s_restarts() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 60.0); + // the miner goes silent (a stopped process, a hung RPC) + assert_eq!(w.tick(149.0, true), Action::None); + let a = w.tick(151.0, true); + assert!(matches!(a, Action::Restart(ref r) if r.contains("no status line")), "{a:?}"); + } + + #[test] + fn no_status_from_the_start() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + assert_eq!(w.tick(89.0, true), Action::None); + assert!(matches!(w.tick(91.0, true), Action::Restart(_))); + } + + #[test] + fn the_miners_own_worker_restart_is_not_doubled() { + // The gfx1036 fault: the miner prints WORKER FAULT, kills the worker, restarts it 2 s later; STATUS lines in + // between say now=0. The app must not restart the miner on top of that. + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 570.0); + w.event(580.0, Event::WorkerRestart(&fault_reason(FAULT_LINE).unwrap())); + let z = parse_status(STATUS_ZERO).unwrap(); + for t in [580.0, 590.0, 600.0, 610.0, 620.0, 630.0, 640.0, 650.0, 660.0] { + w.event(t, Event::Status(&z)); + assert_eq!(w.tick(t, true), Action::None, "the miner owns the restart at {t}"); + } + w.event(665.0, Event::Ready); + healthy(&mut w, 670.0, 900.0); + assert_eq!(w.watchdog_restarts(), 0, "no app restart happened"); + // a worker restart that never comes back: the app steps in after the grace + w.event(910.0, Event::WorkerRestart("no job completed for 60 s with 2 queued")); + for t in (920..=1080).step_by(10) { + w.event(t as f64, Event::Status(&z)); + assert_eq!(w.tick(t as f64, true), Action::None); + } + let a = w.tick(1091.0, true); + assert!(matches!(a, Action::Restart(ref r) if r.contains("did not come back within 180 s")), "{a:?}"); + } + + #[test] + fn exit_43_once_then_faulted() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + healthy(&mut w, 10.0, 60.0); + w.event(70.0, Event::WorkerRestart("cpu re-check: 3 consecutive mismatches (mismatched=3 in this run): the worker computes a wrong program")); + let a = w.event(75.0, Event::Exited(43)); + assert!(matches!(a, Action::Restart(ref r) if r.contains("gave up") && r.contains("wrong program")), "{a:?}"); + w.event(80.0, Event::Started); + let a = w.event(300.0, Event::Exited(43)); + assert!(matches!(a, Action::Fault(_)), "{a:?}"); + // an ordinary crash is the engine's own jittered restart, not the watchdog's + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + assert_eq!(w.event(5.0, Event::Exited(1)), Action::None); + } + + #[test] + fn five_healthy_minutes_renew_the_budget() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + assert!(matches!(w.tick(100.0, true), Action::Restart(_))); + w.event(101.0, Event::Started); + w.event(103.0, Event::Ready); + healthy(&mut w, 110.0, 420.0); + assert_eq!(w.watchdog_restarts(), 0); + // a second incident later is again a restart, not a fault + assert!(matches!(w.tick(520.0, true), Action::Restart(_))); + } + + #[test] + fn a_deliberate_stop_is_not_a_fault() { + let mut w = CardWatch::new(); + w.event(0.0, Event::Started); + w.event(2.0, Event::Ready); + w.event(30.0, Event::Stopped); + assert_eq!(w.tick(500.0, true), Action::None); + assert_eq!(w.event(500.0, Event::Exited(0)), Action::None); + } + + #[test] + fn node_watch_restarts_a_silent_node_with_growing_delay() { + let mut n = NodeWatch::new(); + assert_eq!(n.tick(100.0, true, 119.0, false), None); + assert_eq!(n.tick(100.0, false, 500.0, false), None, "an external node is never restarted"); + assert_eq!(n.tick(100.0, true, 500.0, true), None, "our block was accepted in the last minute: the node is alive"); + assert_eq!(n.tick(100.0, true, 120.0, false), Some(3)); + assert_eq!(n.tick(300.0, true, 120.0, false), Some(12)); + assert_eq!(n.tick(500.0, true, 120.0, false), Some(48)); + assert_eq!(n.restarts_in_window(), 3); + assert_eq!(n.tick(2000.0, true, 120.0, false), Some(3), "the window passed"); + } +} diff --git a/app/igneum-app/ui/app.js b/app/igneum-app/ui/app.js index 9c512dc4..0a11fb57 100644 --- a/app/igneum-app/ui/app.js +++ b/app/igneum-app/ui/app.js @@ -340,12 +340,12 @@ var tb = $('d-cards'); if (!on.length) tb.innerHTML = '
' + (m.cards.length ? 'No card switched on. Open settings to pick one.' : 'No GPU worker. The node runs on its own.') + '
'; else tb.innerHTML = on.map(function (cd) { - var cls = cd.state === 'mining' ? 'mining' : (cd.state === 'failed' || cd.state === 'restarting') ? 'bad' : ''; + var cls = cd.state === 'mining' ? 'mining' : (cd.state === 'failed' || cd.state === 'restarting' || cd.state === 'faulted') ? 'bad' : ''; var word = cd.state === 'restarting' ? ('restart in ' + cd.restart_in_s + ' s') : cd.state; var extra = cd.ids.length ? 'id ' + esc(cd.ids.join(' ')) + '' : ''; return '
' + esc(cd.name) + ' ' + kindWord(cd.kind) + '
' + '
' + cd.hash_now.toFixed(1) + 'MH/s
' + - '
blocks ' + cd.accepted + '' + (cd.rejected ? ' / ' + cd.rejected + ' rejected' : '') + 'avg ' + cd.hash_avg.toFixed(1) + 'identities ' + cd.identities + '' + extra + '
' + + '
blocks ' + cd.accepted + '' + (cd.rejected ? ' / ' + cd.rejected + ' rejected' : '') + 'avg ' + cd.hash_avg.toFixed(1) + 'identities ' + cd.identities + '' + extra + (cd.faults ? 'worker faults ' + cd.faults + '' : '') + (cd.mismatched ? 're-check misses ' + cd.mismatched + '' : '') + '
' + telemetryHtml(cd) + '' + esc(word) + (cd.prepared ? ' · next program ready' : '') + '' + (cd.message ? '
' + esc(cd.message) + '
' : '') + '
'; }).join(''); diff --git a/site/bench.html b/site/bench.html index f038a054..da16e38f 100644 --- a/site/bench.html +++ b/site/bench.html @@ -65,7 +65,7 @@ footer{border-top:1px solid var(--line);padding-block:32px 48px;font-size:13px;c
-
IGNEUM
+
IGNEUM

Engineering log

Every measurement the project has made, newest at the bottom, written by the people and agents who ran it, with the commands and hardware. Prototype numbers are not mining numbers and say so.

diff --git a/site/build.mjs b/site/build.mjs index 626f4a8c..da514cc1 100644 --- a/site/build.mjs +++ b/site/build.mjs @@ -120,7 +120,7 @@ footer{border-top:1px solid var(--line);padding-block:32px 48px;font-size:13px;c
-
${mark}IGNEUM
+
${mark}IGNEUM

${esc(heading)}

${note}

@@ -207,4 +207,31 @@ if (existsSync(join(docs, 'bench-log.md'))) { writeFileSync(jp, JSON.stringify(j, null, 2) + '\n'); built.push('journey.json (' + j.log.length + ' entries)'); } +// the public bench table (/miners): one row per card, generator version and miner version, from site/miner-bench.json +// (the bench-log numbers that exist today, and rows later jobs append); rendered through the same scrub as /bench +{ + const bj = JSON.parse(readFileSync(join(here, 'miner-bench.json'), 'utf8')); + const rows = bj.rows.slice().sort((a, b) => (a.card < b.card ? -1 : a.card > b.card ? 1 : a.generator < b.generator ? -1 : a.generator > b.generator ? 1 : b.mh_s - a.mh_s)); + const fmt = (n) => Number(n).toLocaleString('en-GB', { maximumFractionDigits: 1 }); + const cell = (r) => [ + r.card, r.generator, fmt(r.mh_s), r.mh_per_w == null ? 'not measured' : fmt(r.mh_per_w), r.miner, r.date, r.source, r.by + (r.note ? '. ' + r.note : ''), + ]; + const table = '
' + ['Card', 'Generator', 'Best MH/s', 'MH per watt', 'Miner', 'Date', 'Source', 'Who measured it'].map(h => ``).join('') + '' + + rows.map(r => '' + cell(r).map(c => ``).join('') + '').join('') + '
${h}
${esc(String(c))}
'; + const body = scrubBench([ + '

The table

', + '

One row per card, generator version and miner version. The rate is the best one measured. Integrated GPUs are not listed. Prototype rows are bench numbers from before the devnet and say so in the miner column.

', + table, + '

How a row gets here

', + '

Every row names the engineering log entry or the job it came from. "Measured by the team" means our own hardware and our own log. "Reported by the fleet" means a machine we do not own, read from the status lines its miner uploads.

', + '

MH per watt needs the card\'s power draw during the run. The app reads it on NVIDIA cards through the driver. Rows get the figure when a run records it.

', + '

There is no other Igneum miner to compare with yet, so this table compares cards, not miners.

', + `

Rows: ${rows.length}. Source file: site/miner-bench.json in the repository.

`, + ].join('\n')); + const toc = [{ lvl: 2, t: 'The table', id: 'table' }, { lvl: 2, t: 'How a row gets here', id: 'how' }]; + writeFileSync(join(here, 'miners.html'), page('Igneum GPU bench table', 'Measured Igneum hash rates per GPU: card, generator version, best MH/s, MH per watt where measured, miner version, date and the log entry each number came from.', body, toc, + 'Measured hash rates per card on the Igneum lottery hash, with the generator version, the miner version, the date and the log entry behind each number.', + { path: '/miners', heading: 'GPU bench table' })); + built.push('miners.html (' + rows.length + ' rows)'); +} console.log('built: ' + built.join(', ')); diff --git a/site/evidence.html b/site/evidence.html index d6769f42..4539f4dc 100644 --- a/site/evidence.html +++ b/site/evidence.html @@ -69,7 +69,7 @@ footer{border-top:1px solid var(--line);padding-block:32px 48px;font-size:13px;c
-
IGNEUM
+
IGNEUM

Evidence

Every claim the homepage and the litepaper make, one row each, with one of five labels: designed (a decision, no code), implemented (code with passing test vectors), tested by the team (measured by the project on a named machine, in the engineering log), reproduced externally (a third party ran the published command and got the published result) and reviewed independently (a named outside reviewer published a finding on that version). Nothing on this chain has been reproduced externally or reviewed independently; every row says so. A label belongs to the exact version in the row, and an audit of one version never covers a newer one. The 12-node cloud network of 4 October 2026 is the project's own, so its rows are tested by the team, not reproduced externally.

diff --git a/site/index.html b/site/index.html index e6c9f906..17e1f16f 100644 --- a/site/index.html +++ b/site/index.html @@ -246,6 +246,7 @@ footer .wrap{padding-block:48px 32px} Economics Litepaper Log + Miners Evidence Live Get the miner @@ -485,6 +486,7 @@ footer .wrap{padding-block:48px 32px} Igneum vs RandomX Built on the shoulders Engineering log + GPU bench table
- +
Litepaper · version 0.1

Mined by GPUs.
Proven by fire.

A proof-of-work chain whose miners also prove every block, run Ethereum's apps, and built so no chip can ever take your place.
diff --git a/site/live.html b/site/live.html index 0fea5529..5f17a8a7 100644 --- a/site/live.html +++ b/site/live.html @@ -153,6 +153,7 @@ footer{border-top:1px solid var(--line);padding-block:48px 32px} Economics Litepaper Log + Miners Live Get the miner
@@ -227,6 +228,7 @@ footer{border-top:1px solid var(--line);padding-block:48px 32px} What Igneum does not claim Igneum vs RandomX Engineering log + GPU bench table
diff --git a/tools/reliability/app-run.mjs b/tools/reliability/app-run.mjs index 49f0f6d6..2b9ef375 100755 --- a/tools/reliability/app-run.mjs +++ b/tools/reliability/app-run.mjs @@ -118,7 +118,7 @@ S['own-restart'] = async () => { const st1 = await poll(); verdict('own-restart', [ { ok: !!st0, what: 'the card mined with a rate above 0 on the fake worker' }, - { ok: !!faulted && /worker fault/.test(card(faulted).message || ''), what: `the card showed the WORKER FAULT reason (${JSON.stringify(card(faulted || {}).message)})` }, + { ok: !!faulted && /worker fault|killed by a guard/.test(card(faulted).message || ''), what: `the card showed the worker fault (${JSON.stringify(card(faulted || {}).message)})` }, { ok: !!back, what: 'mining resumed after the miner restarted its own worker' }, { ok: st1 && card(st1).pid === pid0 && card(st1).restarts === r0, what: `the app did not restart the miner (pid ${pid0} -> ${card(st1 || {}).pid}, app restarts ${r0} -> ${card(st1 || {}).restarts})` }, ], { inject_to_fault_on_card: s(tFault - tInject), fault_to_mining_again: s(tBack - tFault) }); diff --git a/tools/reliability/run.mjs b/tools/reliability/run.mjs index 945dc848..5afd8ede 100755 --- a/tools/reliability/run.mjs +++ b/tools/reliability/run.mjs @@ -13,14 +13,14 @@ // fake-fast-stays the same, but the fault persists: three trips at growing delay, then exit 43 // badfound wrong hashes on every found: three consecutive CPU re-check mismatches stop the worker (X21) // silent the worker stops answering: STATUS lines keep coming, the stall guard fires at 60 s -// prepare-flap with epochs every 60 DAA (the block producer), the worker refuses every prepare: at most two +// prepare-flap with epochs every 60 DAA (a 3-thread CPU block producer on easy genesis bits), the worker refuses every prepare: at most two // sends per pair per epoch, 30 s apart (M27) // // A watcher is trusted only once it fires on a known-good and a known-bad case: every scenario states what must // appear AND what must not, and the run fails if the fault detector saw nothing in the scenarios that fault. import { spawn } from 'node:child_process'; -import { mkdirSync, rmSync, writeFileSync, existsSync, appendFileSync } from 'node:fs'; +import { mkdirSync, rmSync, writeFileSync, existsSync, appendFileSync, readFileSync } from 'node:fs'; import { fileURLToPath } from 'node:url'; import { dirname, join } from 'node:path'; @@ -82,7 +82,13 @@ function startMiner(secs, extra = []) { log(`miner pid ${p.pid} log ${f}`); return m; } -const find = (m, re, after = 0) => m.lines.filter(l => l.t >= after && re.test(l.text)); +// stdout only by default: the miner prints WORKER FAULT and the exit line on both streams +const find = (m, re, after = 0, both = false) => m.lines.filter(l => l.t >= after && (both || !l.err) && re.test(l.text)); +async function waitForErr(m, re, timeoutMs, after = 0) { + const end = Date.now() + timeoutMs; + while (Date.now() < end) { const h = find(m, re, after, true); if (h.length) return h[0]; if (m.exit !== null) return null; await sleep(200); } + return null; +} async function waitFor(m, re, timeoutMs, after = 0) { const end = Date.now() + timeoutMs; while (Date.now() < end) { const h = find(m, re, after); if (h.length) return h[0]; if (m.exit !== null) return null; await sleep(200); } @@ -112,12 +118,17 @@ S['slow-first'] = async () => { const faults = find(m, /WORKER FAULT/); const status = find(m, / STATUS '/); const gaps = status.slice(1).map((l, i) => l.t - status[i].t); + const restarted = find(m, /worker restarted/); + const lastFault = faults.length ? faults[faults.length - 1].t : 0; + const healthyAfter = find(m, / STATUS '.*now=(?!0\.00)/, lastFault); verdict('slow-first', [ { ok: !!ready, what: 'worker reported ready' }, - { ok: faults.length === 0, what: `no WORKER FAULT after a slow first interval (saw ${faults.length})` }, + { ok: faults.length <= 1, what: `at most one WORKER FAULT after a slow start (saw ${faults.length}; the old guard tripped on every interval)` }, + { ok: faults.length === restarted.length, what: `every fault was followed by one restart (faults ${faults.length}, restarts ${restarted.length})` }, { ok: status.length >= 7, what: `STATUS lines kept coming (${status.length} in 85 s)` }, { ok: gaps.every(g => g < 25000), what: `no STATUS gap over 25 s (max ${s(Math.max(0, ...gaps))})` }, - ], { status_lines: status.length, faults: faults.length }); + { ok: healthyAfter.length >= 3, what: `healthy STATUS lines after the last fault with no further trip (${healthyAfter.length})` }, + ], { status_lines: status.length, faults: faults.length, restarts: restarted.length, slow_to_fault: faults.length && ready ? s(faults[0].t - ready.t) : 'none', fault_to_restart: faults.length && restarted.length ? s(restarted[0].t - faults[0].t) : 'n/a' }); }; S['fake-fast'] = async () => { @@ -129,7 +140,7 @@ S['fake-fast'] = async () => { setMode('fast'); const fault = await waitFor(m, /WORKER FAULT/, 60000, tInject); if (fault) setMode('ok'); - const killed = await waitFor(m, /worker killed by a guard/, 20000, tInject); + const killed = await waitForErr(m, /worker killed by a guard/, 20000, tInject); const restarted = await waitFor(m, /worker restarted/, 30000, tInject); const ready = await waitFor(m, /worker: ready/, 30000, restarted ? restarted.t : tInject); // the first STATUS after the restart with a rate above 0 @@ -142,8 +153,9 @@ S['fake-fast'] = async () => { await sleep(15000); stop(m); const statusAfterFault = fault ? find(m, / STATUS '.*faults=[1-9]/, fault.t) : []; - const faultsTotal = find(m, /WORKER FAULT/).filter(l => !l.err).length; + const faultsTotal = find(m, /WORKER FAULT/).length; const restarts = find(m, /worker restarted/).length; + if (!restarted) recovered = null; // a rate after the restart only counts once there was a restart verdict('fake-fast', [ { ok: !!fault, what: 'the guard fired on jobs done in 0.3 ms' }, { ok: statusAfterFault.length > 0, what: 'a STATUS line with faults=1 was printed after the trip (the old guard printed none)' }, @@ -168,8 +180,8 @@ S['fake-fast-stays'] = async () => { setMode('fast'); const end = Date.now() + 300000; while (m.exit === null && Date.now() < end) await sleep(500); - const faults = find(m, /WORKER FAULT/).filter(l => !l.err); - const delays = find(m, /restarting it in (\d+) s/).map(l => Number(/restarting it in (\d+) s/.exec(l.text)[1])); + const faults = find(m, /WORKER FAULT/); + const delays = find(m, /restarting it in (\d+) s/, 0, true).map(l => Number(/restarting it in (\d+) s/.exec(l.text)[1])); const gaveUp = find(m, /exiting with code 43/); verdict('fake-fast-stays', [ { ok: faults.length >= 3, what: `three or more faults (${faults.length})` }, @@ -188,7 +200,7 @@ S['badfound'] = async () => { await sleep(15000); const tInject = Date.now(); setMode('badfound'); - const mism = await waitFor(m, /WORKER MISMATCH/, 30000, tInject); + const mism = await waitForErr(m, /WORKER MISMATCH/, 30000, tInject); // stderr const fault = await waitFor(m, /WORKER FAULT cpu re-check: 3 consecutive mismatches/, 60000, tInject); if (fault) setMode('ok'); const restarted = await waitFor(m, /worker restarted/, 30000, tInject); @@ -201,8 +213,9 @@ S['badfound'] = async () => { } await sleep(12000); stop(m); + if (!restarted) recovered = null; const statusMism = find(m, / STATUS '.*mismatched=[1-9]/); - const mismCount = find(m, /WORKER MISMATCH/).length; + const mismCount = find(m, /WORKER MISMATCH/, 0, true).length; verdict('badfound', [ { ok: !!mism, what: 'the CPU re-check rejected the wrong hash' }, { ok: !!fault, what: 'three consecutive mismatches stopped the worker (X21)' }, @@ -235,6 +248,7 @@ S['silent'] = async () => { } await sleep(12000); stop(m); + if (!restarted) recovered = null; const statusDuringSilence = fault ? find(m, / STATUS '/, tInject).filter(l => l.t < fault.t) : []; verdict('silent', [ { ok: !!fault, what: 'the stall guard fired on a worker that stopped answering' }, @@ -250,9 +264,11 @@ S['silent'] = async () => { S['prepare-flap'] = async () => { // a CPU block producer so the DAA moves and epochs turn every 60 blocks (lead 20) - const cpu = spawn(MINER, ['mine', `grpc://127.0.0.1:${BASE}`, '2', '400', 'cpu', '--engine', 'igneum-pow', '--no-vote', '--payout-label', 'cpu'], { stdio: ['ignore', 'ignore', 'ignore'] }); + const cpu = spawn(MINER, ['mine', `grpc://127.0.0.1:${BASE}`, '3', '400', 'cpu', '--engine', 'igneum-pow', '--no-vote', '--payout-label', 'cpu', '--status-secs', '30'], { stdio: ['ignore', 'pipe', 'pipe'] }); started.push(cpu); log(`cpu block producer pid ${cpu.pid}`); + // the block producer's log, for the block count (the node logs at warn level) + const cf = join(SCRATCH, 'cpu-miner.log'); cpu.stdout.on('data', d => appendFileSync(cf, d)); cpu.stderr.on('data', d => appendFileSync(cf, d)); setMode('preparefail'); const m = startMiner(360); await waitFor(m, /worker: ready/, 20000); @@ -260,7 +276,7 @@ S['prepare-flap'] = async () => { stop(m); try { cpu.kill('SIGTERM'); } catch {} const sent = find(m, /PREPARE sent for epoch seed (\S+) day (\d+)/); const held = find(m, /PREPARE held/); - const failed = find(m, /could not prepare/); + const failed = find(m, /could not prepare/, 0, true); // stderr const changes = find(m, /SEED CHANGE/); // per epoch (between SEED CHANGE lines): sends per pair and the smallest gap between any two sends const epochs = []; @@ -279,11 +295,11 @@ S['prepare-flap'] = async () => { { ok: epochs.every(e => e.maxPerPair <= 2), what: `at most two sends per pair per epoch (${epochs.map(e => e.maxPerPair).join(', ')})` }, { ok: gaps.every(g => g >= 29500), what: `every send at least 30 s after the previous (min gap ${gaps.length ? s(Math.min(...gaps)) : 'n/a'})` }, { ok: held.length >= 1, what: `the miner said why it held a prepare (${held.length} lines)` }, - ], { sends: sent.length, held: held.length, seed_changes: changes.length }); + ], { sends: sent.length, held: held.length, seed_changes: changes.length, blocks_by_the_cpu_producer: (() => { try { return (readFileSync(join(SCRATCH, 'cpu-miner.log'), 'utf8').match(/ACCEPTED block/g) || []).length; } catch { return 'n/a'; } })() }); }; const names = ONLY.length ? ONLY : Object.keys(S); -startNode({ IGNEUM_POW_EPOCH_BLOCKS: '60', IGNEUM_POW_EPOCH_LEAD: '20', IGNEUM_DEVNET_GENESIS_BITS: '0x1f010000' }); +startNode({ IGNEUM_POW_EPOCH_BLOCKS: '60', IGNEUM_POW_EPOCH_LEAD: '20', IGNEUM_DEVNET_GENESIS_BITS: '0x1f100000' }); await sleep(4000); for (const n of names) { if (!S[n]) { log(`unknown scenario ${n}`); continue; } From 89e21a40cde977d35968940e08efc3fe98188194 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Sun, 4 Oct 2026 22:22:39 +0000 Subject: [PATCH 3/3] Site: journey feed picks up the reliability measurement entry Co-Authored-By: Claude Fable 5.1 --- site/journey.json | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/site/journey.json b/site/journey.json index 90394925..58784e02 100644 --- a/site/journey.json +++ b/site/journey.json @@ -50,6 +50,11 @@ } ], "log": [ + { + "date": "2026-10-04", + "text": "Miner fault guards and the app watchdog measured against a fake worker", + "short": "Miner fault guards and the app watchdog measured against a fake worker" + }, { "date": "2026-10-04", "text": "Shard proving on the RTX 5090: a full shard compressed in 10.9 s, a two-shard block aggregated in 2.2 s, all verified", @@ -175,6 +180,11 @@ "text": "Devnet-v4 integration: nine branches merged, 3-node test network on the merged node, Windows cross-build", "short": "Devnet v4: nine branches merged into one node" }, + { + "date": "2026-10-03", + "text": "Difficulty controller: devnet record, simulator, Igneum dual-lane rule, 3-node CPU test network", + "short": "Igneum dual-lane difficulty rule built and simulated" + }, { "date": "2026-10-03", "text": "First devnet blocks on the real lottery hash: CPU, then Metal GPU, three worker implementations", @@ -239,16 +249,6 @@ "date": "2026-10-03", "text": "Windows node package: igneumd cross-compiled for x86_64-pc-windows-gnu, two-peer sync test", "short": "Windows node package: cross-compiled, two-peer sync" - }, - { - "date": "2026-10-03", - "text": "R3.26 / M15: PoW checked after the cheap checks, cache-build cap, attack before and after", - "short": "Cache-build attack closed: 10.6 s of rebuilds to 14 ms" - }, - { - "date": "2026-10-03", - "text": "Difficulty controller: devnet record, simulator, Igneum dual-lane rule, 3-node CPU test network", - "short": "Igneum dual-lane difficulty rule built and simulated" } ] }